aboutsummaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
authorDanilo M. <danix@danix.xyz>2026-09-18 20:23:23 +0200
committerDanilo M. <danix@danix.xyz>2026-09-18 20:23:23 +0200
commit108031af8ec10411fdba9af4cc3ef2703c60c7de (patch)
tree71fef9ef0531d2c3c662f5052293485e86d9c677
parentdae64bcaf9c9a8a8eda134297e5844efc1f4c9c0 (diff)
downloadllamachat-108031af8ec10411fdba9af4cc3ef2703c60c7de.tar.gz
llamachat-108031af8ec10411fdba9af4cc3ef2703c60c7de.zip
feat: audio model capability in metadata and dialog
-rw-r--r--llamachat/modeldialog.py37
-rw-r--r--llamachat/models.py30
-rw-r--r--llamachat/providers.py3
-rwxr-xr-xtest_llamachat.py47
4 files changed, 103 insertions, 14 deletions
diff --git a/llamachat/modeldialog.py b/llamachat/modeldialog.py
index ebebf4a..c6f7501 100644
--- a/llamachat/modeldialog.py
+++ b/llamachat/modeldialog.py
@@ -50,16 +50,19 @@ def _number(text: str, cast):
return value
-def to_info(ctx_text, vision, in_text, out_text, vision_prefill=None):
+def to_info(
+ ctx_text, vision, in_text, out_text,
+ vision_prefill=None, audio=False, audio_prefill=None,
+):
"""Build a ModelInfo from the dialog's raw field values.
- Vision is the one tri-state field, and a binary checkbox cannot hold
- three states on its own. The checkbox carries the user's intent (checked
- means "yes"), and vision_prefill carries what was known before the
+ Vision and audio are the tri-state fields, and a binary checkbox cannot
+ hold three states on its own. The checkbox carries the user's intent
+ (checked means "yes"), and the prefill carries what was known before the
dialog opened: an unchecked box over an unknown prefill stays unknown
- rather than writing False, which would shadow a provider-level
- vision=True through models.resolve's pick(). An unchecked box over a
- known prefill, True or False, is a real "no".
+ rather than writing False, which would shadow a provider-level True
+ through models.resolve's pick(). An unchecked box over a known prefill,
+ True or False, is a real "no".
"""
if vision:
resolved_vision = True
@@ -67,21 +70,31 @@ def to_info(ctx_text, vision, in_text, out_text, vision_prefill=None):
resolved_vision = None
else:
resolved_vision = False
+
+ if audio:
+ resolved_audio = True
+ elif audio_prefill is None:
+ resolved_audio = None
+ else:
+ resolved_audio = False
+
return ModelInfo(
ctx_size=_number(ctx_text, int),
vision=resolved_vision,
+ audio=resolved_audio,
price_in=_number(in_text, float),
price_out=_number(out_text, float),
)
-def to_fields(info: ModelInfo) -> tuple[str, bool, str, str]:
+def to_fields(info: ModelInfo) -> tuple[str, bool, str, str, bool]:
"""The inverse, for prefilling. Unknown becomes an empty field."""
return (
"" if info.ctx_size is None else str(info.ctx_size),
bool(info.vision),
"" if info.price_in is None else str(info.price_in),
"" if info.price_out is None else str(info.price_out),
+ bool(info.audio),
)
@@ -97,12 +110,15 @@ class ModelDialog(QDialog):
self.setWindowTitle("Model settings")
self.model_id = model_id
self._vision_prefill = info.vision
+ self._audio_prefill = info.audio
- ctx, vision, price_in, price_out = to_fields(info)
+ ctx, vision, price_in, price_out, audio = to_fields(info)
self.ctx = QLineEdit(ctx)
self.ctx.setPlaceholderText("unknown")
self.vision = QCheckBox("Accepts images")
self.vision.setChecked(vision)
+ self.audio = QCheckBox("Accepts audio")
+ self.audio.setChecked(audio)
self.price_in = QLineEdit(price_in)
self.price_in.setPlaceholderText("unpriced")
self.price_out = QLineEdit(price_out)
@@ -116,6 +132,7 @@ class ModelDialog(QDialog):
form = QFormLayout()
form.addRow("Context size (tokens)", self.ctx)
form.addRow("", self.vision)
+ form.addRow("", self.audio)
form.addRow("Input price (per 1M tokens)", self.price_in)
form.addRow("Output price (per 1M tokens)", self.price_out)
layout.addLayout(form)
@@ -142,4 +159,6 @@ class ModelDialog(QDialog):
self.price_in.text(),
self.price_out.text(),
vision_prefill=self._vision_prefill,
+ audio=self.audio.isChecked(),
+ audio_prefill=self._audio_prefill,
) \ No newline at end of file
diff --git a/llamachat/models.py b/llamachat/models.py
index 2b2da2b..e2f7922 100644
--- a/llamachat/models.py
+++ b/llamachat/models.py
@@ -16,7 +16,7 @@
import configparser
import math
import os
-from dataclasses import dataclass
+from dataclasses import dataclass, replace
from pathlib import Path
from . import providers as providers_mod
@@ -46,13 +46,20 @@ class ModelInfo:
ctx_size: int | None = None
vision: bool | None = None
+ audio: bool | None = None
price_in: float | None = None
price_out: float | None = None
def is_empty(self) -> bool:
return all(
v is None
- for v in (self.ctx_size, self.vision, self.price_in, self.price_out)
+ for v in (
+ self.ctx_size,
+ self.vision,
+ self.audio,
+ self.price_in,
+ self.price_out,
+ )
)
@@ -140,6 +147,7 @@ class ModelStore:
info = ModelInfo(
ctx_size=_get(section, "ctx_size", int),
vision=_get_bool(section, "vision"),
+ audio=_get_bool(section, "audio"),
price_in=_get(section, "price_in", float),
price_out=_get(section, "price_out", float),
)
@@ -154,6 +162,7 @@ class ModelStore:
for key, value in (
("ctx_size", info.ctx_size),
("vision", info.vision),
+ ("audio", info.audio),
("price_in", info.price_in),
("price_out", info.price_out),
):
@@ -238,11 +247,28 @@ def resolve(model_id: str, table: dict, store: "ModelStore") -> ModelInfo:
return ModelInfo(
ctx_size=pick(stored.ctx_size, provider.ctx_size),
vision=pick(stored.vision, provider.vision),
+ audio=pick(stored.audio, provider.audio),
price_in=pick(stored.price_in, provider.price_in),
price_out=pick(stored.price_out, provider.price_out),
)
+def apply_inputs(info: ModelInfo, inputs) -> ModelInfo:
+ """Overlay router-reported input modalities on a ModelInfo.
+
+ `inputs` is the set from the server's `architecture.input_modalities`, or
+ None when the server reports none. A reported modality is a fact, so it
+ overrides a stored or provider guess for vision and audio at once.
+ """
+ if inputs is None:
+ return info
+ return replace(
+ info,
+ vision="image" in inputs,
+ audio="audio" in inputs,
+ )
+
+
def is_billable(model_id: str, table: dict) -> bool:
"""Whether this model costs money, regardless of prices being known.
diff --git a/llamachat/providers.py b/llamachat/providers.py
index 23ad081..db8efc5 100644
--- a/llamachat/providers.py
+++ b/llamachat/providers.py
@@ -44,6 +44,7 @@ class Provider:
# None rather than 0: unset must stay distinguishable from "zero".
ctx_size: int | None = None
vision: bool | None = None
+ audio: bool | None = None
price_in: float | None = None
price_out: float | None = None
# Provider-specific reasoning token budget. Currently SiliconFlow only;
@@ -185,6 +186,7 @@ def parse(values: dict, warnings: list[str] | None = None) -> dict[str, Provider
if not isinstance(needles, (list, tuple)):
needles = [needles]
vision = entry.get("vision")
+ audio = entry.get("audio")
out[name] = Provider(
name=name,
base_url=base_url,
@@ -192,6 +194,7 @@ def parse(values: dict, warnings: list[str] | None = None) -> dict[str, Provider
filter=[str(f) for f in needles],
ctx_size=_number(entry.get("ctx_size"), int),
vision=None if vision is None else bool(vision),
+ audio=None if audio is None else bool(audio),
price_in=_number(entry.get("price_in"), float),
price_out=_number(entry.get("price_out"), float),
thinking_budget=_number(entry.get("thinking_budget"), int),
diff --git a/test_llamachat.py b/test_llamachat.py
index 44f569e..faa0a00 100755
--- a/test_llamachat.py
+++ b/test_llamachat.py
@@ -1256,6 +1256,7 @@ def test_provider_parsing():
"price_out": 0.9,
"thinking_budget": 8192,
"replay_reasoning": True,
+ "audio": True,
},
}
}
@@ -1273,6 +1274,9 @@ def test_provider_parsing():
# Off by default: replaying reasoning costs context and input tokens.
assert parsed["local"].replay_reasoning is False
+ assert parsed["together"].audio is True
+ assert parsed["local"].audio is None
+
# An old config: bare base_url, no providers table at all.
legacy = providers.parse({"base_url": "http://localhost:8181"})
assert set(legacy) == {"local"}
@@ -1827,6 +1831,13 @@ def test_models_store():
assert models.ModelStore(path).get("together:novision").vision is False
assert "vision = false" in path.read_text(encoding="utf-8")
+ # Audio is the same tri-state and must survive the same round trip.
+ again.save("together:hear", models.ModelInfo(audio=True))
+ assert models.ModelStore(path).get("together:hear").audio is True
+ assert "audio = true" in path.read_text(encoding="utf-8")
+ assert models.ModelInfo(audio=True).is_empty() is False
+ assert models.ModelInfo(audio=False).is_empty() is False
+
# Cancelling a dialog over a model we already know must not erase it,
# nor mark it skipped: the marker means "no real keys", so a section
# holding both would be a state no reader is written to expect.
@@ -1887,6 +1898,23 @@ def test_models_store():
print("ok models.ini storage")
+def test_audio_metadata_layers():
+ """Audio resolves store over provider; router overlay is Task 4's job."""
+ from llamachat import models, providers
+
+ with tempfile.TemporaryDirectory() as tmp:
+ table = providers.parse(
+ {"providers": {"local": {"base_url": "http://x", "audio": True}}}
+ )
+ store = models.ModelStore(Path(tmp) / "models.ini")
+
+ assert models.resolve("m", table, store).audio is True
+ store.save("m", models.ModelInfo(audio=False))
+ # The store is more specific than the provider.
+ assert models.resolve("m", table, store).audio is False
+ print("ok audio metadata layers")
+
+
def test_metadata_and_cost():
"""models.ini beats provider defaults beats unknown; cost sums per model."""
from llamachat import models, providers
@@ -2583,11 +2611,23 @@ def test_model_dialog_values():
ctx_text="", vision=False, in_text="", out_text="", vision_prefill=False
).vision is False
+ # Audio is tri-state in the same way.
+ assert modeldialog.to_info(
+ ctx_text="", vision=False, in_text="", out_text="", audio=True
+ ).audio is True
+ assert modeldialog.to_info(
+ ctx_text="", vision=False, in_text="", out_text="", audio=False
+ ).audio is None
+ assert modeldialog.to_info(
+ ctx_text="", vision=False, in_text="", out_text="",
+ audio=False, audio_prefill=True,
+ ).audio is False
+
# Prefill is the inverse: unknown becomes an empty field.
- assert modeldialog.to_fields(models.ModelInfo()) == ("", False, "", "")
+ assert modeldialog.to_fields(models.ModelInfo()) == ("", False, "", "", False)
assert modeldialog.to_fields(
- models.ModelInfo(ctx_size=4096, vision=True, price_in=0.5)
- ) == ("4096", True, "0.5", "")
+ models.ModelInfo(ctx_size=4096, vision=True, price_in=0.5, audio=True)
+ ) == ("4096", True, "0.5", "", True)
print("ok model dialog value conversion")
@@ -3944,6 +3984,7 @@ if __name__ == "__main__":
test_config_providers()
test_replays_reasoning()
test_models_store()
+ test_audio_metadata_layers()
test_metadata_and_cost()
test_token_column_migration()
test_usage_columns()