aboutsummaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
authorDanilo M. <danix@danix.xyz>2026-09-19 10:44:48 +0200
committerDanilo M. <danix@danix.xyz>2026-09-19 10:44:48 +0200
commitc10c750c83a3cde13f7ca574dac673e2358df522 (patch)
tree11d3694da330091155e9c066399ce39c790000fc
parent309c0dd8b0fc294f67af8a82b78830a7610b700f (diff)
downloadllamachat-c10c750c83a3cde13f7ca574dac673e2358df522.tar.gz
llamachat-c10c750c83a3cde13f7ca574dac673e2358df522.zip
feat: icon-only toolbar and audio input follow-upsHEADmaster
Make the chat toolbar icon-only (tooltips retained) and add an icon_size config key, default 28, so the drawn glyphs can be scaled. Qt was drawing the 24px pixmaps at its ~16px default; the explicit icon size fixes that. Fix the deferred audio minors: AudioRecorder.start() now halts a prior source and checks the source error, so a failed capture no longer claims to be recording. Send now validates every pending modality in one gate, so a mixed image+audio clip cannot reach a model without vision. Sub-second clips no longer name voice-note-0s.wav, and non-WAV audio containers classify as unknown, guarded by a test.
-rw-r--r--CHANGELOG.md18
-rw-r--r--llamachat/audio.py10
-rw-r--r--llamachat/backend.py4
-rw-r--r--llamachat/config.py8
-rw-r--r--llamachat/modeldialog.py2
-rw-r--r--llamachat/models.py3
-rw-r--r--llamachat/ui.py147
-rwxr-xr-xtest_llamachat.py5
8 files changed, 144 insertions, 53 deletions
diff --git a/CHANGELOG.md b/CHANGELOG.md
index a2a3635..c685d54 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -69,10 +69,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
moment once used. The text comes from the source markdown, since Qt emits
one `<pre>` per line and the block cannot be recovered from the rendered
HTML.
-- Icons on the chat buttons (New, Attach, Clear files, Stop, Send), drawn
- at runtime so there is still no icon file to install. Send is now a square
- button beside the entry box with its label under the icon, and New/Attach
- sit at the right end of the picker row.
+- Icons on the chat buttons (New, Attach, Clear files, Stop, Send, Record),
+ drawn at runtime so there is still no icon file to install. The toolbar
+ buttons are icon-only, each with a tooltip, and `icon_size` in config.toml
+ sets the glyph size.
- Skills. Instruction files read from `~/.agents/skills/*/SKILL.md` can be
loaded into the model's context on demand: the model calls a `load_skill`
tool, or you type `/skill-name` or mention a skill's name. Loaded skills
@@ -118,6 +118,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
assistant tool-call message. Interleaved-thinking models (DeepSeek V3.2+/V4,
GLM-4.7+ on SiliconFlow) require this and stop answering — ending the turn at
the thinking — when it is dropped.
+- Audio capture no longer claims to be recording when it is not.
+ `AudioRecorder.start()` halts any previous source, checks the new source's
+ error, and reports a generic failure; a second start without a stop no
+ longer leaks the first source.
+- A pending image attached before a model switch could reach a model without
+ vision when sent alongside audio. The send-time gate now checks every
+ pending modality at once and offers a single switch to a model that accepts
+ them all.
+- A sub-second recording could be named `voice-note-0s.wav`; the history
+ marker now rounds to a minimum of one second.
Tool calling and reasoning output vary between providers. Web search on a
cloud model may not work as reliably as it does with a local router.
diff --git a/llamachat/audio.py b/llamachat/audio.py
index 67b568c..58c7dd6 100644
--- a/llamachat/audio.py
+++ b/llamachat/audio.py
@@ -80,10 +80,13 @@ if AVAILABLE:
self._timer.timeout.connect(self._on_cap)
def start(self) -> bool:
- """Begin capture. False when there is no input device."""
+ """Begin capture. False when there is no input device or it fails."""
device = QMediaDevices.defaultAudioInput()
if device.isNull():
return False
+ # A second start() without a stop() would leak the previous source
+ # and leave two captures feeding two timers.
+ self._halt()
fmt = QAudioFormat()
fmt.setSampleRate(RATE)
fmt.setChannelCount(CHANNELS)
@@ -93,7 +96,10 @@ if AVAILABLE:
self._io = QBuffer(self._buffer)
self._io.open(QIODevice.WriteOnly)
self._source = QAudioSource(device, fmt, self)
- self._source.start(self._io)
+ error = self._source.start(self._io)
+ if error is not None:
+ self._halt()
+ return False
self._timer.start(MAX_SECONDS * 1000)
return True
diff --git a/llamachat/backend.py b/llamachat/backend.py
index 8b9115c..cfdecde 100644
--- a/llamachat/backend.py
+++ b/llamachat/backend.py
@@ -104,7 +104,7 @@ class SkillsConfig:
class Attachment:
"""A file the user attached, ready for both the API and the database."""
path: Path
- kind: str # 'text' | 'image'
+ kind: str # 'text' | 'image' | 'audio'
mime: str
size: int
sha256: str
@@ -188,7 +188,7 @@ def build_user_content(text: str, attachments: list[Attachment]):
"""Assemble one OpenAI-format user message from text plus attachments.
Returns a plain string when there is nothing but text, and the
- multi-part content array when images are involved.
+ multi-part content array when images or audio are involved.
"""
parts: list[str] = []
for att in attachments:
diff --git a/llamachat/config.py b/llamachat/config.py
index f7dab5d..cfcda57 100644
--- a/llamachat/config.py
+++ b/llamachat/config.py
@@ -36,6 +36,9 @@ DEFAULTS = {
"attach_ctx_fraction": 0.5,
# Rough chars-per-token used to turn a ctx-size into a char budget.
"chars_per_token": 3.5,
+ # Toolbar button glyph size in pixels. The icons are drawn, not loaded,
+ # so this is the only knob for their on-screen size.
+ "icon_size": 28,
# Named prompt selected for new conversations. Empty means the global
# default.md, and "none" means no system prompt at all.
"default_prompt": "",
@@ -86,6 +89,7 @@ class Config:
request_timeout: int
attach_ctx_fraction: float
chars_per_token: float
+ icon_size: int
default_prompt: str
prompts_dir: Path
state_path: Path
@@ -157,6 +161,7 @@ def load(path: Path = CONFIG_PATH) -> Config:
request_timeout=int(values["request_timeout"]),
attach_ctx_fraction=float(values["attach_ctx_fraction"]),
chars_per_token=float(values["chars_per_token"]),
+ icon_size=int(values["icon_size"]),
default_prompt=str(values["default_prompt"]),
prompts_dir=path.parent / "prompts",
# Window layout, remembered between runs. Not user-editable config,
@@ -211,6 +216,9 @@ def write_default(path: Path = CONFIG_PATH) -> Path:
f'attach_ctx_fraction = {DEFAULTS["attach_ctx_fraction"]}\n'
f'chars_per_token = {DEFAULTS["chars_per_token"]}\n'
'\n'
+ '# Toolbar button icon size in pixels.\n'
+ f'icon_size = {DEFAULTS["icon_size"]}\n'
+ '\n'
'# Web search through SearXNG. The model decides when to search; it\n'
'# is offered a web_search tool and calls it when a question needs\n'
'# current information. Both search_enabled and search_url are\n'
diff --git a/llamachat/modeldialog.py b/llamachat/modeldialog.py
index c6f7501..a29a1d8 100644
--- a/llamachat/modeldialog.py
+++ b/llamachat/modeldialog.py
@@ -99,7 +99,7 @@ def to_fields(info: ModelInfo) -> tuple[str, bool, str, str, bool]:
class ModelDialog(QDialog):
- """Context size, vision and prices for one model.
+ """Context size, vision, audio and prices for one model.
Prefilled from the provider's defaults, so the common case is checking
the numbers rather than typing them.
diff --git a/llamachat/models.py b/llamachat/models.py
index e2f7922..fb9d9fd 100644
--- a/llamachat/models.py
+++ b/llamachat/models.py
@@ -170,7 +170,8 @@ class ModelStore:
section.pop(key, None)
elif isinstance(value, bool):
# Before the numeric branch: bool is a subclass of int, so
- # str(False) would otherwise write "False" for vision.
+ # str(False) would otherwise write "False" for vision and
+ # audio.
section[key] = "true" if value else "false"
else:
section[key] = str(value)
diff --git a/llamachat/ui.py b/llamachat/ui.py
index 91321e4..b45daad 100644
--- a/llamachat/ui.py
+++ b/llamachat/ui.py
@@ -22,7 +22,8 @@ import re
from pathlib import Path
from PySide6.QtCore import (
- QObject, QPointF, QRect, QRectF, QSettings, Qt, QThread, QTimer, Signal, Slot,
+ QObject, QPointF, QRect, QRectF, QSettings, QSize, Qt, QThread, QTimer,
+ Signal, Slot,
)
from PySide6.QtGui import (
QAction, QColor, QDesktopServices, QIcon, QKeySequence, QPainter,
@@ -58,12 +59,17 @@ COPY_SCHEME = "x-llamachat-copy:"
# The buttons get their glyphs drawn at runtime, so there is no icon file
# to install. All the marks share the brand blue, which also makes each
# button an unambiguous control rather than a bare text label.
-def _button_icon(kind: str) -> QIcon:
- """A small monochrome glyph for one of the chat buttons."""
- pixmap = QPixmap(24, 24)
+def _button_icon(kind: str, size: int = 24) -> QIcon:
+ """A small monochrome glyph for one of the chat buttons.
+
+ The glyph coordinates below are authored on a 24x24 grid and the painter
+ is scaled, so one call renders crisply at any configured icon size.
+ """
+ pixmap = QPixmap(size, size)
pixmap.fill(QColor(0, 0, 0, 0))
painter = QPainter(pixmap)
painter.setRenderHint(QPainter.Antialiasing)
+ painter.scale(size / 24.0, size / 24.0)
colour = QColor("#3F5EFB")
pen = QPen(colour, 2.2)
pen.setCapStyle(Qt.RoundCap)
@@ -124,6 +130,16 @@ def _button_icon(kind: str) -> QIcon:
return QIcon(pixmap)
+def _set_button_icon(button, kind: str, size: int) -> None:
+ """Give a toolbar button its glyph and match the icon box to it.
+
+ Without the explicit icon size, Qt draws the pixmap at the style's small
+ default (about 16px) regardless of how large it was rendered.
+ """
+ button.setIcon(_button_icon(kind, size))
+ button.setIconSize(QSize(size, size))
+
+
class ContextMeter(QWidget):
"""A bar showing how much of the model's context the next request uses.
@@ -694,10 +710,10 @@ class ChatWindow(QMainWindow):
compose.addWidget(self.input, 1)
# A square send control beside the entry box, like a chat app.
+ size = self.cfg.icon_size
self.send_button = QToolButton()
- self.send_button.setIcon(_button_icon("send"))
- self.send_button.setText("Send")
- self.send_button.setToolButtonStyle(Qt.ToolButtonTextUnderIcon)
+ _set_button_icon(self.send_button, "send", size)
+ self.send_button.setToolButtonStyle(Qt.ToolButtonIconOnly)
self.send_button.setFixedSize(44, 44)
self.send_button.setToolTip("Send (Ctrl+Enter)")
self.send_button.clicked.connect(self.send)
@@ -750,33 +766,36 @@ class ChatWindow(QMainWindow):
buttons.addWidget(self.prompt_button)
buttons.addStretch()
- self.clear_attach_button = QPushButton("Clear files")
- self.clear_attach_button.setIcon(_button_icon("clear"))
+ self.clear_attach_button = QPushButton()
+ _set_button_icon(self.clear_attach_button, "clear", size)
+ self.clear_attach_button.setToolTip("Clear attached files")
self.clear_attach_button.clicked.connect(self.clear_attachments)
self.clear_attach_button.hide()
buttons.addWidget(self.clear_attach_button)
- self.stop_button = QPushButton("Stop")
- self.stop_button.setIcon(_button_icon("stop"))
+ self.stop_button = QPushButton()
+ _set_button_icon(self.stop_button, "stop", size)
+ self.stop_button.setToolTip("Stop generating")
self.stop_button.clicked.connect(self.stop_stream)
self.stop_button.hide()
buttons.addWidget(self.stop_button)
- self.new_button = QPushButton("New")
- self.new_button.setIcon(_button_icon("new"))
+ self.new_button = QPushButton()
+ _set_button_icon(self.new_button, "new", size)
self.new_button.setToolTip("Start a fresh conversation (Ctrl+N)")
self.new_button.clicked.connect(self.new_session)
buttons.addWidget(self.new_button)
- self.record_button = QPushButton("Record")
- self.record_button.setIcon(_button_icon("record"))
+ self.record_button = QPushButton()
+ _set_button_icon(self.record_button, "record", size)
self.record_button.setToolTip("Record audio for an audio-capable model")
self.record_button.clicked.connect(self._toggle_record)
self.record_button.hide()
buttons.addWidget(self.record_button)
- self.attach_button = QPushButton("Attach…")
- self.attach_button.setIcon(_button_icon("attach"))
+ self.attach_button = QPushButton()
+ _set_button_icon(self.attach_button, "attach", size)
+ self.attach_button.setToolTip("Attach files")
self.attach_button.clicked.connect(self._on_attach_clicked)
buttons.addWidget(self.attach_button)
@@ -994,39 +1013,73 @@ class ChatWindow(QMainWindow):
names.append(name)
return names
- def audio_models(self) -> list[str]:
+ def _modality_models(self, audio: bool, vision: bool) -> list[str]:
+ """Model ids that satisfy every required modality.
+
+ Audio is strict, vision lenient: only a model that reports audio can
+ receive a recording, while an unknown vision status gets the benefit
+ of the doubt. Same asymmetry as the attach-time gate.
+ """
names = []
for i in range(self.model_box.count()):
name = self.model_box.itemData(i)
- if self.model_info(name).audio is True:
- names.append(name)
+ info = self.model_info(name)
+ if audio and info.audio is not True:
+ continue
+ if vision and info.vision is False:
+ continue
+ names.append(name)
return names
- def _ensure_audio_model(self) -> bool:
- """Offer to switch to an audio-capable model. True when one is active.
+ def _ensure_modalities(self, *, audio: bool, vision: bool) -> bool:
+ """Offer a single switch to a model that accepts everything pending.
- Unlike vision, an unknown model is not given the benefit of the
- doubt: audio input is rare and explicitly reported, so only a model
- known to accept it is allowed to receive a recording.
+ Returns True when the current model can receive the attachments. One
+ combined switch, rather than a vision switch then an audio switch,
+ stops a mixed clip ping-ponging between a vision-only and an
+ audio-only model.
"""
- if self.current_info().audio is True:
+ info = self.current_info()
+ if (not audio or info.audio is True) and (
+ not vision or info.vision is not False
+ ):
return True
- candidates = self.audio_models()
+ candidates = self._modality_models(audio, vision)
if not candidates:
- QMessageBox.warning(
- self,
- "No audio model",
- "No model on the router reports audio input, so a recording "
- "cannot be sent.",
- )
+ if audio and vision:
+ title = "No compatible model"
+ body = (
+ "No model accepts both audio and images, so this clip "
+ "cannot be sent."
+ )
+ elif audio:
+ title = "No audio model"
+ body = (
+ "No model on the router reports audio input, so a "
+ "recording cannot be sent."
+ )
+ else:
+ title = "No vision model"
+ body = (
+ "No model on the router has an mmproj file configured, so "
+ "images cannot be sent."
+ )
+ QMessageBox.warning(self, title, body)
return False
+ if audio and vision:
+ need = "Audio and images need a model that accepts both."
+ elif audio:
+ need = "Audio needs a model that accepts voice input."
+ else:
+ need = "Images need a vision-capable model."
+
target = candidates[0]
answer = QMessageBox.question(
self,
"Switch model?",
- f"Audio needs a model that accepts voice input.\n\n"
+ f"{need}\n\n"
f"Switch to {target}?\n\n"
"The router unloads the current model to do this, so the next "
"reply will take several seconds to start.",
@@ -1267,9 +1320,15 @@ class ChatWindow(QMainWindow):
self.recorder is not None and self.current_info().audio is True
)
self.record_button.setVisible(can)
- self.record_button.setText("Stop" if self.recording else "Record")
- self.record_button.setIcon(
- _button_icon("stop" if self.recording else "record")
+ _set_button_icon(
+ self.record_button,
+ "stop" if self.recording else "record",
+ self.cfg.icon_size,
+ )
+ self.record_button.setToolTip(
+ "Stop recording"
+ if self.recording
+ else "Record audio for an audio-capable model"
)
def _toggle_record(self) -> None:
@@ -1280,7 +1339,7 @@ class ChatWindow(QMainWindow):
if self.recorder is None or self.current_info().audio is not True:
return
if not self.recorder.start():
- self.show_status("No microphone available.", error=True)
+ self.show_status("Could not start recording.", error=True)
return
self.recording = True
self.show_status("Recording. Click Stop when done.")
@@ -1306,7 +1365,7 @@ class ChatWindow(QMainWindow):
else:
self.hide_status()
self.attachments.append(
- backend.audio_attachment(audio.to_wav(pcm), round(seconds))
+ backend.audio_attachment(audio.to_wav(pcm), max(1, round(seconds)))
)
self.update_attach_label()
@@ -1346,12 +1405,14 @@ class ChatWindow(QMainWindow):
if not text and not self.attachments:
return
- if any(a.kind == "audio" for a in self.attachments):
+ needs_audio = any(a.kind == "audio" for a in self.attachments)
+ needs_vision = any(a.kind == "image" for a in self.attachments)
+ if needs_audio or needs_vision:
# A switch here changes the model, so this must run before the
- # model is read below.
- if not self._ensure_audio_model():
+ # model is read below. One combined gate covers a mixed clip.
+ if not self._ensure_modalities(audio=needs_audio, vision=needs_vision):
return
- if not text:
+ if needs_audio and not text:
text = backend.AUDIO_PROMPT
model = self.current_model()
diff --git a/test_llamachat.py b/test_llamachat.py
index bf43c77..f665ee3 100755
--- a/test_llamachat.py
+++ b/test_llamachat.py
@@ -251,6 +251,11 @@ def test_classify():
assert backend.classify(Path("a.so")) == "unknown"
assert backend.classify(Path("a.wav")) == "audio"
assert backend.classify(Path("a.WAV")) == "audio"
+ # Only WAV is accepted; other audio containers must not be labelled
+ # audio, since the request hardcodes format "wav".
+ assert backend.classify(Path("a.mp3")) == "unknown"
+ assert backend.classify(Path("a.ogg")) == "unknown"
+ assert backend.classify(Path("a.flac")) == "unknown"
print("ok file classification")