aboutsummaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
authorDanilo M. <danix@danix.xyz>2026-09-18 20:42:10 +0200
committerDanilo M. <danix@danix.xyz>2026-09-18 20:42:10 +0200
commit309c0dd8b0fc294f67af8a82b78830a7610b700f (patch)
treed090c70e2ef1534dfd31b194d078e9f305d70cbe
parentddd2715e2c49d02db7ece5b0ae32c589cfd1e4db (diff)
downloadllamachat-309c0dd8b0fc294f67af8a82b78830a7610b700f.tar.gz
llamachat-309c0dd8b0fc294f67af8a82b78830a7610b700f.zip
feat: mark audio-capable models in the picker
-rw-r--r--CHANGELOG.md6
-rw-r--r--README.md5
-rw-r--r--llamachat/ui.py15
-rwxr-xr-xtest_llamachat.py26
4 files changed, 45 insertions, 7 deletions
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 9eb1ee5..a2a3635 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -84,8 +84,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
selected model reports audio input (`architecture.input_modalities` from the
router, or the per-model "Accepts audio" setting); the clip is captured at
16 kHz mono, sent as an `input_audio` content part, and discarded after
- send, leaving only a `voice-note Ns` marker in history. Capture needs
- QtMultimedia from PySide6-Addons and is absent without it.
+ send, leaving only a `voice-note Ns` marker in history. Audio-capable
+ models show a microphone (🎙) in the picker beside their name, as vision
+ models show an eye. Capture needs QtMultimedia from PySide6-Addons and is
+ absent without it.
### Fixed
diff --git a/README.md b/README.md
index b1ad066..dadc43d 100644
--- a/README.md
+++ b/README.md
@@ -519,8 +519,9 @@ short default instruction so the request always carries a text part.
Capability comes from the router: llama-server's model router reports each
model's accepted inputs in `/v1/models` (`architecture.input_modalities`), and
-a model listing `"audio"` there gets the control. Cloud models, whose
-endpoints do not report modalities, can be marked by hand with the
+a model listing `"audio"` there gets the control and a microphone (🎙) beside
+its name in the picker, the way a vision model gets an eye. Cloud models,
+whose endpoints do not report modalities, can be marked by hand with the
**Accepts audio** box in the model settings dialog.
Recording is 16 kHz mono WAV, the format speech encoders expect. The clip is
diff --git a/llamachat/ui.py b/llamachat/ui.py
index d5c6d34..91321e4 100644
--- a/llamachat/ui.py
+++ b/llamachat/ui.py
@@ -929,8 +929,7 @@ class ChatWindow(QMainWindow):
self.model_box.clear()
for name in available:
info = self.model_info(name)
- label = f"{name} 👁" if info.vision else name
- self.model_box.addItem(label, name)
+ self.model_box.addItem(_model_label(name, info), name)
self.model_box.blockSignals(False)
target = previous or self.cfg.default_model
@@ -2305,7 +2304,17 @@ def _column(row, name: str, default: str = "") -> str:
def _strip_marker(name: str) -> str:
- return name.replace(" 👁", "").strip()
+ return name.replace(" 👁", "").replace(" 🎙", "").strip()
+
+
+def _model_label(name: str, info) -> str:
+ """The picker label: the model id plus a mark for each input modality.
+
+ The marks are display only; the item's data is always the bare id, and
+ `_strip_marker` is their inverse.
+ """
+ marks = (" 👁" if info.vision else "") + (" 🎙" if info.audio else "")
+ return name + marks
def _plain(text: str) -> str:
diff --git a/test_llamachat.py b/test_llamachat.py
index 20aea57..bf43c77 100755
--- a/test_llamachat.py
+++ b/test_llamachat.py
@@ -520,6 +520,31 @@ def test_audio_bubble_marker():
print("ok audio bubble marker")
+def test_model_label_markers():
+ """The picker marks vision and audio, and strips both back off."""
+ import os
+
+ os.environ.setdefault("QT_QPA_PLATFORM", "offscreen")
+
+ from llamachat import models
+ from llamachat.ui import _model_label, _strip_marker
+
+ assert _model_label("m", models.ModelInfo()) == "m"
+ assert _model_label("m", models.ModelInfo(vision=True)) == "m 👁"
+ assert _model_label("m", models.ModelInfo(audio=True)) == "m 🎙"
+ assert (
+ _model_label("m", models.ModelInfo(vision=True, audio=True))
+ == "m 👁 🎙"
+ )
+
+ # The strip is the inverse, and leaves an unmarked name alone.
+ assert _strip_marker("m 👁 🎙") == "m"
+ assert _strip_marker("m 🎙") == "m"
+ assert _strip_marker("m 👁") == "m"
+ assert _strip_marker("m") == "m"
+ print("ok model label markers")
+
+
def test_unbalanced_backticks():
"""An odd backtick cannot reflow the rest of a reply as code."""
import os
@@ -4065,6 +4090,7 @@ if __name__ == "__main__":
test_prompt_column_migration()
test_markdown_rendering()
test_audio_bubble_marker()
+ test_model_label_markers()
test_unbalanced_backticks()
test_sidebar_toggle()
test_shortcuts()