diff options
| author | Danilo M. <danix@danix.xyz> | 2026-09-19 10:44:48 +0200 |
|---|---|---|
| committer | Danilo M. <danix@danix.xyz> | 2026-09-19 10:44:48 +0200 |
| commit | c10c750c83a3cde13f7ca574dac673e2358df522 (patch) | |
| tree | 11d3694da330091155e9c066399ce39c790000fc | |
| parent | 309c0dd8b0fc294f67af8a82b78830a7610b700f (diff) | |
| download | llamachat-c10c750c83a3cde13f7ca574dac673e2358df522.tar.gz llamachat-c10c750c83a3cde13f7ca574dac673e2358df522.zip | |
Make the chat toolbar icon-only (tooltips retained) and add an icon_size config key, default 28, so the drawn glyphs can be scaled. Qt was drawing the 24px pixmaps at its ~16px default; the explicit icon size fixes that.
Fix the deferred audio minors: AudioRecorder.start() now halts a prior source and checks the source error, so a failed capture no longer claims to be recording. Send now validates every pending modality in one gate, so a mixed image+audio clip cannot reach a model without vision. Sub-second clips no longer name voice-note-0s.wav, and non-WAV audio containers classify as unknown, guarded by a test.
| -rw-r--r-- | CHANGELOG.md | 18 | ||||
| -rw-r--r-- | llamachat/audio.py | 10 | ||||
| -rw-r--r-- | llamachat/backend.py | 4 | ||||
| -rw-r--r-- | llamachat/config.py | 8 | ||||
| -rw-r--r-- | llamachat/modeldialog.py | 2 | ||||
| -rw-r--r-- | llamachat/models.py | 3 | ||||
| -rw-r--r-- | llamachat/ui.py | 147 | ||||
| -rwxr-xr-x | test_llamachat.py | 5 |
8 files changed, 144 insertions, 53 deletions
diff --git a/CHANGELOG.md b/CHANGELOG.md index a2a3635..c685d54 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -69,10 +69,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 moment once used. The text comes from the source markdown, since Qt emits one `<pre>` per line and the block cannot be recovered from the rendered HTML. -- Icons on the chat buttons (New, Attach, Clear files, Stop, Send), drawn - at runtime so there is still no icon file to install. Send is now a square - button beside the entry box with its label under the icon, and New/Attach - sit at the right end of the picker row. +- Icons on the chat buttons (New, Attach, Clear files, Stop, Send, Record), + drawn at runtime so there is still no icon file to install. The toolbar + buttons are icon-only, each with a tooltip, and `icon_size` in config.toml + sets the glyph size. - Skills. Instruction files read from `~/.agents/skills/*/SKILL.md` can be loaded into the model's context on demand: the model calls a `load_skill` tool, or you type `/skill-name` or mention a skill's name. Loaded skills @@ -118,6 +118,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 assistant tool-call message. Interleaved-thinking models (DeepSeek V3.2+/V4, GLM-4.7+ on SiliconFlow) require this and stop answering — ending the turn at the thinking — when it is dropped. +- Audio capture no longer claims to be recording when it is not. + `AudioRecorder.start()` halts any previous source, checks the new source's + error, and reports a generic failure; a second start without a stop no + longer leaks the first source. +- A pending image attached before a model switch could reach a model without + vision when sent alongside audio. The send-time gate now checks every + pending modality at once and offers a single switch to a model that accepts + them all. +- A sub-second recording could be named `voice-note-0s.wav`; the history + marker now rounds to a minimum of one second. Tool calling and reasoning output vary between providers. Web search on a cloud model may not work as reliably as it does with a local router. diff --git a/llamachat/audio.py b/llamachat/audio.py index 67b568c..58c7dd6 100644 --- a/llamachat/audio.py +++ b/llamachat/audio.py @@ -80,10 +80,13 @@ if AVAILABLE: self._timer.timeout.connect(self._on_cap) def start(self) -> bool: - """Begin capture. False when there is no input device.""" + """Begin capture. False when there is no input device or it fails.""" device = QMediaDevices.defaultAudioInput() if device.isNull(): return False + # A second start() without a stop() would leak the previous source + # and leave two captures feeding two timers. + self._halt() fmt = QAudioFormat() fmt.setSampleRate(RATE) fmt.setChannelCount(CHANNELS) @@ -93,7 +96,10 @@ if AVAILABLE: self._io = QBuffer(self._buffer) self._io.open(QIODevice.WriteOnly) self._source = QAudioSource(device, fmt, self) - self._source.start(self._io) + error = self._source.start(self._io) + if error is not None: + self._halt() + return False self._timer.start(MAX_SECONDS * 1000) return True diff --git a/llamachat/backend.py b/llamachat/backend.py index 8b9115c..cfdecde 100644 --- a/llamachat/backend.py +++ b/llamachat/backend.py @@ -104,7 +104,7 @@ class SkillsConfig: class Attachment: """A file the user attached, ready for both the API and the database.""" path: Path - kind: str # 'text' | 'image' + kind: str # 'text' | 'image' | 'audio' mime: str size: int sha256: str @@ -188,7 +188,7 @@ def build_user_content(text: str, attachments: list[Attachment]): """Assemble one OpenAI-format user message from text plus attachments. Returns a plain string when there is nothing but text, and the - multi-part content array when images are involved. + multi-part content array when images or audio are involved. """ parts: list[str] = [] for att in attachments: diff --git a/llamachat/config.py b/llamachat/config.py index f7dab5d..cfcda57 100644 --- a/llamachat/config.py +++ b/llamachat/config.py @@ -36,6 +36,9 @@ DEFAULTS = { "attach_ctx_fraction": 0.5, # Rough chars-per-token used to turn a ctx-size into a char budget. "chars_per_token": 3.5, + # Toolbar button glyph size in pixels. The icons are drawn, not loaded, + # so this is the only knob for their on-screen size. + "icon_size": 28, # Named prompt selected for new conversations. Empty means the global # default.md, and "none" means no system prompt at all. "default_prompt": "", @@ -86,6 +89,7 @@ class Config: request_timeout: int attach_ctx_fraction: float chars_per_token: float + icon_size: int default_prompt: str prompts_dir: Path state_path: Path @@ -157,6 +161,7 @@ def load(path: Path = CONFIG_PATH) -> Config: request_timeout=int(values["request_timeout"]), attach_ctx_fraction=float(values["attach_ctx_fraction"]), chars_per_token=float(values["chars_per_token"]), + icon_size=int(values["icon_size"]), default_prompt=str(values["default_prompt"]), prompts_dir=path.parent / "prompts", # Window layout, remembered between runs. Not user-editable config, @@ -211,6 +216,9 @@ def write_default(path: Path = CONFIG_PATH) -> Path: f'attach_ctx_fraction = {DEFAULTS["attach_ctx_fraction"]}\n' f'chars_per_token = {DEFAULTS["chars_per_token"]}\n' '\n' + '# Toolbar button icon size in pixels.\n' + f'icon_size = {DEFAULTS["icon_size"]}\n' + '\n' '# Web search through SearXNG. The model decides when to search; it\n' '# is offered a web_search tool and calls it when a question needs\n' '# current information. Both search_enabled and search_url are\n' diff --git a/llamachat/modeldialog.py b/llamachat/modeldialog.py index c6f7501..a29a1d8 100644 --- a/llamachat/modeldialog.py +++ b/llamachat/modeldialog.py @@ -99,7 +99,7 @@ def to_fields(info: ModelInfo) -> tuple[str, bool, str, str, bool]: class ModelDialog(QDialog): - """Context size, vision and prices for one model. + """Context size, vision, audio and prices for one model. Prefilled from the provider's defaults, so the common case is checking the numbers rather than typing them. diff --git a/llamachat/models.py b/llamachat/models.py index e2f7922..fb9d9fd 100644 --- a/llamachat/models.py +++ b/llamachat/models.py @@ -170,7 +170,8 @@ class ModelStore: section.pop(key, None) elif isinstance(value, bool): # Before the numeric branch: bool is a subclass of int, so - # str(False) would otherwise write "False" for vision. + # str(False) would otherwise write "False" for vision and + # audio. section[key] = "true" if value else "false" else: section[key] = str(value) diff --git a/llamachat/ui.py b/llamachat/ui.py index 91321e4..b45daad 100644 --- a/llamachat/ui.py +++ b/llamachat/ui.py @@ -22,7 +22,8 @@ import re from pathlib import Path from PySide6.QtCore import ( - QObject, QPointF, QRect, QRectF, QSettings, Qt, QThread, QTimer, Signal, Slot, + QObject, QPointF, QRect, QRectF, QSettings, QSize, Qt, QThread, QTimer, + Signal, Slot, ) from PySide6.QtGui import ( QAction, QColor, QDesktopServices, QIcon, QKeySequence, QPainter, @@ -58,12 +59,17 @@ COPY_SCHEME = "x-llamachat-copy:" # The buttons get their glyphs drawn at runtime, so there is no icon file # to install. All the marks share the brand blue, which also makes each # button an unambiguous control rather than a bare text label. -def _button_icon(kind: str) -> QIcon: - """A small monochrome glyph for one of the chat buttons.""" - pixmap = QPixmap(24, 24) +def _button_icon(kind: str, size: int = 24) -> QIcon: + """A small monochrome glyph for one of the chat buttons. + + The glyph coordinates below are authored on a 24x24 grid and the painter + is scaled, so one call renders crisply at any configured icon size. + """ + pixmap = QPixmap(size, size) pixmap.fill(QColor(0, 0, 0, 0)) painter = QPainter(pixmap) painter.setRenderHint(QPainter.Antialiasing) + painter.scale(size / 24.0, size / 24.0) colour = QColor("#3F5EFB") pen = QPen(colour, 2.2) pen.setCapStyle(Qt.RoundCap) @@ -124,6 +130,16 @@ def _button_icon(kind: str) -> QIcon: return QIcon(pixmap) +def _set_button_icon(button, kind: str, size: int) -> None: + """Give a toolbar button its glyph and match the icon box to it. + + Without the explicit icon size, Qt draws the pixmap at the style's small + default (about 16px) regardless of how large it was rendered. + """ + button.setIcon(_button_icon(kind, size)) + button.setIconSize(QSize(size, size)) + + class ContextMeter(QWidget): """A bar showing how much of the model's context the next request uses. @@ -694,10 +710,10 @@ class ChatWindow(QMainWindow): compose.addWidget(self.input, 1) # A square send control beside the entry box, like a chat app. + size = self.cfg.icon_size self.send_button = QToolButton() - self.send_button.setIcon(_button_icon("send")) - self.send_button.setText("Send") - self.send_button.setToolButtonStyle(Qt.ToolButtonTextUnderIcon) + _set_button_icon(self.send_button, "send", size) + self.send_button.setToolButtonStyle(Qt.ToolButtonIconOnly) self.send_button.setFixedSize(44, 44) self.send_button.setToolTip("Send (Ctrl+Enter)") self.send_button.clicked.connect(self.send) @@ -750,33 +766,36 @@ class ChatWindow(QMainWindow): buttons.addWidget(self.prompt_button) buttons.addStretch() - self.clear_attach_button = QPushButton("Clear files") - self.clear_attach_button.setIcon(_button_icon("clear")) + self.clear_attach_button = QPushButton() + _set_button_icon(self.clear_attach_button, "clear", size) + self.clear_attach_button.setToolTip("Clear attached files") self.clear_attach_button.clicked.connect(self.clear_attachments) self.clear_attach_button.hide() buttons.addWidget(self.clear_attach_button) - self.stop_button = QPushButton("Stop") - self.stop_button.setIcon(_button_icon("stop")) + self.stop_button = QPushButton() + _set_button_icon(self.stop_button, "stop", size) + self.stop_button.setToolTip("Stop generating") self.stop_button.clicked.connect(self.stop_stream) self.stop_button.hide() buttons.addWidget(self.stop_button) - self.new_button = QPushButton("New") - self.new_button.setIcon(_button_icon("new")) + self.new_button = QPushButton() + _set_button_icon(self.new_button, "new", size) self.new_button.setToolTip("Start a fresh conversation (Ctrl+N)") self.new_button.clicked.connect(self.new_session) buttons.addWidget(self.new_button) - self.record_button = QPushButton("Record") - self.record_button.setIcon(_button_icon("record")) + self.record_button = QPushButton() + _set_button_icon(self.record_button, "record", size) self.record_button.setToolTip("Record audio for an audio-capable model") self.record_button.clicked.connect(self._toggle_record) self.record_button.hide() buttons.addWidget(self.record_button) - self.attach_button = QPushButton("Attach…") - self.attach_button.setIcon(_button_icon("attach")) + self.attach_button = QPushButton() + _set_button_icon(self.attach_button, "attach", size) + self.attach_button.setToolTip("Attach files") self.attach_button.clicked.connect(self._on_attach_clicked) buttons.addWidget(self.attach_button) @@ -994,39 +1013,73 @@ class ChatWindow(QMainWindow): names.append(name) return names - def audio_models(self) -> list[str]: + def _modality_models(self, audio: bool, vision: bool) -> list[str]: + """Model ids that satisfy every required modality. + + Audio is strict, vision lenient: only a model that reports audio can + receive a recording, while an unknown vision status gets the benefit + of the doubt. Same asymmetry as the attach-time gate. + """ names = [] for i in range(self.model_box.count()): name = self.model_box.itemData(i) - if self.model_info(name).audio is True: - names.append(name) + info = self.model_info(name) + if audio and info.audio is not True: + continue + if vision and info.vision is False: + continue + names.append(name) return names - def _ensure_audio_model(self) -> bool: - """Offer to switch to an audio-capable model. True when one is active. + def _ensure_modalities(self, *, audio: bool, vision: bool) -> bool: + """Offer a single switch to a model that accepts everything pending. - Unlike vision, an unknown model is not given the benefit of the - doubt: audio input is rare and explicitly reported, so only a model - known to accept it is allowed to receive a recording. + Returns True when the current model can receive the attachments. One + combined switch, rather than a vision switch then an audio switch, + stops a mixed clip ping-ponging between a vision-only and an + audio-only model. """ - if self.current_info().audio is True: + info = self.current_info() + if (not audio or info.audio is True) and ( + not vision or info.vision is not False + ): return True - candidates = self.audio_models() + candidates = self._modality_models(audio, vision) if not candidates: - QMessageBox.warning( - self, - "No audio model", - "No model on the router reports audio input, so a recording " - "cannot be sent.", - ) + if audio and vision: + title = "No compatible model" + body = ( + "No model accepts both audio and images, so this clip " + "cannot be sent." + ) + elif audio: + title = "No audio model" + body = ( + "No model on the router reports audio input, so a " + "recording cannot be sent." + ) + else: + title = "No vision model" + body = ( + "No model on the router has an mmproj file configured, so " + "images cannot be sent." + ) + QMessageBox.warning(self, title, body) return False + if audio and vision: + need = "Audio and images need a model that accepts both." + elif audio: + need = "Audio needs a model that accepts voice input." + else: + need = "Images need a vision-capable model." + target = candidates[0] answer = QMessageBox.question( self, "Switch model?", - f"Audio needs a model that accepts voice input.\n\n" + f"{need}\n\n" f"Switch to {target}?\n\n" "The router unloads the current model to do this, so the next " "reply will take several seconds to start.", @@ -1267,9 +1320,15 @@ class ChatWindow(QMainWindow): self.recorder is not None and self.current_info().audio is True ) self.record_button.setVisible(can) - self.record_button.setText("Stop" if self.recording else "Record") - self.record_button.setIcon( - _button_icon("stop" if self.recording else "record") + _set_button_icon( + self.record_button, + "stop" if self.recording else "record", + self.cfg.icon_size, + ) + self.record_button.setToolTip( + "Stop recording" + if self.recording + else "Record audio for an audio-capable model" ) def _toggle_record(self) -> None: @@ -1280,7 +1339,7 @@ class ChatWindow(QMainWindow): if self.recorder is None or self.current_info().audio is not True: return if not self.recorder.start(): - self.show_status("No microphone available.", error=True) + self.show_status("Could not start recording.", error=True) return self.recording = True self.show_status("Recording. Click Stop when done.") @@ -1306,7 +1365,7 @@ class ChatWindow(QMainWindow): else: self.hide_status() self.attachments.append( - backend.audio_attachment(audio.to_wav(pcm), round(seconds)) + backend.audio_attachment(audio.to_wav(pcm), max(1, round(seconds))) ) self.update_attach_label() @@ -1346,12 +1405,14 @@ class ChatWindow(QMainWindow): if not text and not self.attachments: return - if any(a.kind == "audio" for a in self.attachments): + needs_audio = any(a.kind == "audio" for a in self.attachments) + needs_vision = any(a.kind == "image" for a in self.attachments) + if needs_audio or needs_vision: # A switch here changes the model, so this must run before the - # model is read below. - if not self._ensure_audio_model(): + # model is read below. One combined gate covers a mixed clip. + if not self._ensure_modalities(audio=needs_audio, vision=needs_vision): return - if not text: + if needs_audio and not text: text = backend.AUDIO_PROMPT model = self.current_model() diff --git a/test_llamachat.py b/test_llamachat.py index bf43c77..f665ee3 100755 --- a/test_llamachat.py +++ b/test_llamachat.py @@ -251,6 +251,11 @@ def test_classify(): assert backend.classify(Path("a.so")) == "unknown" assert backend.classify(Path("a.wav")) == "audio" assert backend.classify(Path("a.WAV")) == "audio" + # Only WAV is accepted; other audio containers must not be labelled + # audio, since the request hardcodes format "wav". + assert backend.classify(Path("a.mp3")) == "unknown" + assert backend.classify(Path("a.ogg")) == "unknown" + assert backend.classify(Path("a.flac")) == "unknown" print("ok file classification") |
