diff options
| author | Danilo M. <danix@danix.xyz> | 2026-09-18 20:42:10 +0200 |
|---|---|---|
| committer | Danilo M. <danix@danix.xyz> | 2026-09-18 20:42:10 +0200 |
| commit | 309c0dd8b0fc294f67af8a82b78830a7610b700f (patch) | |
| tree | d090c70e2ef1534dfd31b194d078e9f305d70cbe | |
| parent | ddd2715e2c49d02db7ece5b0ae32c589cfd1e4db (diff) | |
| download | llamachat-309c0dd8b0fc294f67af8a82b78830a7610b700f.tar.gz llamachat-309c0dd8b0fc294f67af8a82b78830a7610b700f.zip | |
feat: mark audio-capable models in the picker
| -rw-r--r-- | CHANGELOG.md | 6 | ||||
| -rw-r--r-- | README.md | 5 | ||||
| -rw-r--r-- | llamachat/ui.py | 15 | ||||
| -rwxr-xr-x | test_llamachat.py | 26 |
4 files changed, 45 insertions, 7 deletions
diff --git a/CHANGELOG.md b/CHANGELOG.md index 9eb1ee5..a2a3635 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -84,8 +84,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 selected model reports audio input (`architecture.input_modalities` from the router, or the per-model "Accepts audio" setting); the clip is captured at 16 kHz mono, sent as an `input_audio` content part, and discarded after - send, leaving only a `voice-note Ns` marker in history. Capture needs - QtMultimedia from PySide6-Addons and is absent without it. + send, leaving only a `voice-note Ns` marker in history. Audio-capable + models show a microphone (🎙) in the picker beside their name, as vision + models show an eye. Capture needs QtMultimedia from PySide6-Addons and is + absent without it. ### Fixed @@ -519,8 +519,9 @@ short default instruction so the request always carries a text part. Capability comes from the router: llama-server's model router reports each model's accepted inputs in `/v1/models` (`architecture.input_modalities`), and -a model listing `"audio"` there gets the control. Cloud models, whose -endpoints do not report modalities, can be marked by hand with the +a model listing `"audio"` there gets the control and a microphone (🎙) beside +its name in the picker, the way a vision model gets an eye. Cloud models, +whose endpoints do not report modalities, can be marked by hand with the **Accepts audio** box in the model settings dialog. Recording is 16 kHz mono WAV, the format speech encoders expect. The clip is diff --git a/llamachat/ui.py b/llamachat/ui.py index d5c6d34..91321e4 100644 --- a/llamachat/ui.py +++ b/llamachat/ui.py @@ -929,8 +929,7 @@ class ChatWindow(QMainWindow): self.model_box.clear() for name in available: info = self.model_info(name) - label = f"{name} 👁" if info.vision else name - self.model_box.addItem(label, name) + self.model_box.addItem(_model_label(name, info), name) self.model_box.blockSignals(False) target = previous or self.cfg.default_model @@ -2305,7 +2304,17 @@ def _column(row, name: str, default: str = "") -> str: def _strip_marker(name: str) -> str: - return name.replace(" 👁", "").strip() + return name.replace(" 👁", "").replace(" 🎙", "").strip() + + +def _model_label(name: str, info) -> str: + """The picker label: the model id plus a mark for each input modality. + + The marks are display only; the item's data is always the bare id, and + `_strip_marker` is their inverse. + """ + marks = (" 👁" if info.vision else "") + (" 🎙" if info.audio else "") + return name + marks def _plain(text: str) -> str: diff --git a/test_llamachat.py b/test_llamachat.py index 20aea57..bf43c77 100755 --- a/test_llamachat.py +++ b/test_llamachat.py @@ -520,6 +520,31 @@ def test_audio_bubble_marker(): print("ok audio bubble marker") +def test_model_label_markers(): + """The picker marks vision and audio, and strips both back off.""" + import os + + os.environ.setdefault("QT_QPA_PLATFORM", "offscreen") + + from llamachat import models + from llamachat.ui import _model_label, _strip_marker + + assert _model_label("m", models.ModelInfo()) == "m" + assert _model_label("m", models.ModelInfo(vision=True)) == "m 👁" + assert _model_label("m", models.ModelInfo(audio=True)) == "m 🎙" + assert ( + _model_label("m", models.ModelInfo(vision=True, audio=True)) + == "m 👁 🎙" + ) + + # The strip is the inverse, and leaves an unmarked name alone. + assert _strip_marker("m 👁 🎙") == "m" + assert _strip_marker("m 🎙") == "m" + assert _strip_marker("m 👁") == "m" + assert _strip_marker("m") == "m" + print("ok model label markers") + + def test_unbalanced_backticks(): """An odd backtick cannot reflow the rest of a reply as code.""" import os @@ -4065,6 +4090,7 @@ if __name__ == "__main__": test_prompt_column_migration() test_markdown_rendering() test_audio_bubble_marker() + test_model_label_markers() test_unbalanced_backticks() test_sidebar_toggle() test_shortcuts() |
