diff options
| author | Danilo M. <danix@danix.xyz> | 2026-09-18 20:23:23 +0200 |
|---|---|---|
| committer | Danilo M. <danix@danix.xyz> | 2026-09-18 20:23:23 +0200 |
| commit | 108031af8ec10411fdba9af4cc3ef2703c60c7de (patch) | |
| tree | 71fef9ef0531d2c3c662f5052293485e86d9c677 | |
| parent | dae64bcaf9c9a8a8eda134297e5844efc1f4c9c0 (diff) | |
| download | llamachat-108031af8ec10411fdba9af4cc3ef2703c60c7de.tar.gz llamachat-108031af8ec10411fdba9af4cc3ef2703c60c7de.zip | |
feat: audio model capability in metadata and dialog
| -rw-r--r-- | llamachat/modeldialog.py | 37 | ||||
| -rw-r--r-- | llamachat/models.py | 30 | ||||
| -rw-r--r-- | llamachat/providers.py | 3 | ||||
| -rwxr-xr-x | test_llamachat.py | 47 |
4 files changed, 103 insertions, 14 deletions
diff --git a/llamachat/modeldialog.py b/llamachat/modeldialog.py index ebebf4a..c6f7501 100644 --- a/llamachat/modeldialog.py +++ b/llamachat/modeldialog.py @@ -50,16 +50,19 @@ def _number(text: str, cast): return value -def to_info(ctx_text, vision, in_text, out_text, vision_prefill=None): +def to_info( + ctx_text, vision, in_text, out_text, + vision_prefill=None, audio=False, audio_prefill=None, +): """Build a ModelInfo from the dialog's raw field values. - Vision is the one tri-state field, and a binary checkbox cannot hold - three states on its own. The checkbox carries the user's intent (checked - means "yes"), and vision_prefill carries what was known before the + Vision and audio are the tri-state fields, and a binary checkbox cannot + hold three states on its own. The checkbox carries the user's intent + (checked means "yes"), and the prefill carries what was known before the dialog opened: an unchecked box over an unknown prefill stays unknown - rather than writing False, which would shadow a provider-level - vision=True through models.resolve's pick(). An unchecked box over a - known prefill, True or False, is a real "no". + rather than writing False, which would shadow a provider-level True + through models.resolve's pick(). An unchecked box over a known prefill, + True or False, is a real "no". """ if vision: resolved_vision = True @@ -67,21 +70,31 @@ def to_info(ctx_text, vision, in_text, out_text, vision_prefill=None): resolved_vision = None else: resolved_vision = False + + if audio: + resolved_audio = True + elif audio_prefill is None: + resolved_audio = None + else: + resolved_audio = False + return ModelInfo( ctx_size=_number(ctx_text, int), vision=resolved_vision, + audio=resolved_audio, price_in=_number(in_text, float), price_out=_number(out_text, float), ) -def to_fields(info: ModelInfo) -> tuple[str, bool, str, str]: +def to_fields(info: ModelInfo) -> tuple[str, bool, str, str, bool]: """The inverse, for prefilling. Unknown becomes an empty field.""" return ( "" if info.ctx_size is None else str(info.ctx_size), bool(info.vision), "" if info.price_in is None else str(info.price_in), "" if info.price_out is None else str(info.price_out), + bool(info.audio), ) @@ -97,12 +110,15 @@ class ModelDialog(QDialog): self.setWindowTitle("Model settings") self.model_id = model_id self._vision_prefill = info.vision + self._audio_prefill = info.audio - ctx, vision, price_in, price_out = to_fields(info) + ctx, vision, price_in, price_out, audio = to_fields(info) self.ctx = QLineEdit(ctx) self.ctx.setPlaceholderText("unknown") self.vision = QCheckBox("Accepts images") self.vision.setChecked(vision) + self.audio = QCheckBox("Accepts audio") + self.audio.setChecked(audio) self.price_in = QLineEdit(price_in) self.price_in.setPlaceholderText("unpriced") self.price_out = QLineEdit(price_out) @@ -116,6 +132,7 @@ class ModelDialog(QDialog): form = QFormLayout() form.addRow("Context size (tokens)", self.ctx) form.addRow("", self.vision) + form.addRow("", self.audio) form.addRow("Input price (per 1M tokens)", self.price_in) form.addRow("Output price (per 1M tokens)", self.price_out) layout.addLayout(form) @@ -142,4 +159,6 @@ class ModelDialog(QDialog): self.price_in.text(), self.price_out.text(), vision_prefill=self._vision_prefill, + audio=self.audio.isChecked(), + audio_prefill=self._audio_prefill, )
\ No newline at end of file diff --git a/llamachat/models.py b/llamachat/models.py index 2b2da2b..e2f7922 100644 --- a/llamachat/models.py +++ b/llamachat/models.py @@ -16,7 +16,7 @@ import configparser import math import os -from dataclasses import dataclass +from dataclasses import dataclass, replace from pathlib import Path from . import providers as providers_mod @@ -46,13 +46,20 @@ class ModelInfo: ctx_size: int | None = None vision: bool | None = None + audio: bool | None = None price_in: float | None = None price_out: float | None = None def is_empty(self) -> bool: return all( v is None - for v in (self.ctx_size, self.vision, self.price_in, self.price_out) + for v in ( + self.ctx_size, + self.vision, + self.audio, + self.price_in, + self.price_out, + ) ) @@ -140,6 +147,7 @@ class ModelStore: info = ModelInfo( ctx_size=_get(section, "ctx_size", int), vision=_get_bool(section, "vision"), + audio=_get_bool(section, "audio"), price_in=_get(section, "price_in", float), price_out=_get(section, "price_out", float), ) @@ -154,6 +162,7 @@ class ModelStore: for key, value in ( ("ctx_size", info.ctx_size), ("vision", info.vision), + ("audio", info.audio), ("price_in", info.price_in), ("price_out", info.price_out), ): @@ -238,11 +247,28 @@ def resolve(model_id: str, table: dict, store: "ModelStore") -> ModelInfo: return ModelInfo( ctx_size=pick(stored.ctx_size, provider.ctx_size), vision=pick(stored.vision, provider.vision), + audio=pick(stored.audio, provider.audio), price_in=pick(stored.price_in, provider.price_in), price_out=pick(stored.price_out, provider.price_out), ) +def apply_inputs(info: ModelInfo, inputs) -> ModelInfo: + """Overlay router-reported input modalities on a ModelInfo. + + `inputs` is the set from the server's `architecture.input_modalities`, or + None when the server reports none. A reported modality is a fact, so it + overrides a stored or provider guess for vision and audio at once. + """ + if inputs is None: + return info + return replace( + info, + vision="image" in inputs, + audio="audio" in inputs, + ) + + def is_billable(model_id: str, table: dict) -> bool: """Whether this model costs money, regardless of prices being known. diff --git a/llamachat/providers.py b/llamachat/providers.py index 23ad081..db8efc5 100644 --- a/llamachat/providers.py +++ b/llamachat/providers.py @@ -44,6 +44,7 @@ class Provider: # None rather than 0: unset must stay distinguishable from "zero". ctx_size: int | None = None vision: bool | None = None + audio: bool | None = None price_in: float | None = None price_out: float | None = None # Provider-specific reasoning token budget. Currently SiliconFlow only; @@ -185,6 +186,7 @@ def parse(values: dict, warnings: list[str] | None = None) -> dict[str, Provider if not isinstance(needles, (list, tuple)): needles = [needles] vision = entry.get("vision") + audio = entry.get("audio") out[name] = Provider( name=name, base_url=base_url, @@ -192,6 +194,7 @@ def parse(values: dict, warnings: list[str] | None = None) -> dict[str, Provider filter=[str(f) for f in needles], ctx_size=_number(entry.get("ctx_size"), int), vision=None if vision is None else bool(vision), + audio=None if audio is None else bool(audio), price_in=_number(entry.get("price_in"), float), price_out=_number(entry.get("price_out"), float), thinking_budget=_number(entry.get("thinking_budget"), int), diff --git a/test_llamachat.py b/test_llamachat.py index 44f569e..faa0a00 100755 --- a/test_llamachat.py +++ b/test_llamachat.py @@ -1256,6 +1256,7 @@ def test_provider_parsing(): "price_out": 0.9, "thinking_budget": 8192, "replay_reasoning": True, + "audio": True, }, } } @@ -1273,6 +1274,9 @@ def test_provider_parsing(): # Off by default: replaying reasoning costs context and input tokens. assert parsed["local"].replay_reasoning is False + assert parsed["together"].audio is True + assert parsed["local"].audio is None + # An old config: bare base_url, no providers table at all. legacy = providers.parse({"base_url": "http://localhost:8181"}) assert set(legacy) == {"local"} @@ -1827,6 +1831,13 @@ def test_models_store(): assert models.ModelStore(path).get("together:novision").vision is False assert "vision = false" in path.read_text(encoding="utf-8") + # Audio is the same tri-state and must survive the same round trip. + again.save("together:hear", models.ModelInfo(audio=True)) + assert models.ModelStore(path).get("together:hear").audio is True + assert "audio = true" in path.read_text(encoding="utf-8") + assert models.ModelInfo(audio=True).is_empty() is False + assert models.ModelInfo(audio=False).is_empty() is False + # Cancelling a dialog over a model we already know must not erase it, # nor mark it skipped: the marker means "no real keys", so a section # holding both would be a state no reader is written to expect. @@ -1887,6 +1898,23 @@ def test_models_store(): print("ok models.ini storage") +def test_audio_metadata_layers(): + """Audio resolves store over provider; router overlay is Task 4's job.""" + from llamachat import models, providers + + with tempfile.TemporaryDirectory() as tmp: + table = providers.parse( + {"providers": {"local": {"base_url": "http://x", "audio": True}}} + ) + store = models.ModelStore(Path(tmp) / "models.ini") + + assert models.resolve("m", table, store).audio is True + store.save("m", models.ModelInfo(audio=False)) + # The store is more specific than the provider. + assert models.resolve("m", table, store).audio is False + print("ok audio metadata layers") + + def test_metadata_and_cost(): """models.ini beats provider defaults beats unknown; cost sums per model.""" from llamachat import models, providers @@ -2583,11 +2611,23 @@ def test_model_dialog_values(): ctx_text="", vision=False, in_text="", out_text="", vision_prefill=False ).vision is False + # Audio is tri-state in the same way. + assert modeldialog.to_info( + ctx_text="", vision=False, in_text="", out_text="", audio=True + ).audio is True + assert modeldialog.to_info( + ctx_text="", vision=False, in_text="", out_text="", audio=False + ).audio is None + assert modeldialog.to_info( + ctx_text="", vision=False, in_text="", out_text="", + audio=False, audio_prefill=True, + ).audio is False + # Prefill is the inverse: unknown becomes an empty field. - assert modeldialog.to_fields(models.ModelInfo()) == ("", False, "", "") + assert modeldialog.to_fields(models.ModelInfo()) == ("", False, "", "", False) assert modeldialog.to_fields( - models.ModelInfo(ctx_size=4096, vision=True, price_in=0.5) - ) == ("4096", True, "0.5", "") + models.ModelInfo(ctx_size=4096, vision=True, price_in=0.5, audio=True) + ) == ("4096", True, "0.5", "", True) print("ok model dialog value conversion") @@ -3944,6 +3984,7 @@ if __name__ == "__main__": test_config_providers() test_replays_reasoning() test_models_store() + test_audio_metadata_layers() test_metadata_and_cost() test_token_column_migration() test_usage_columns() |
