aboutsummaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
-rw-r--r--llamachat/audio.py116
-rwxr-xr-xtest_llamachat.py26
2 files changed, 142 insertions, 0 deletions
diff --git a/llamachat/audio.py b/llamachat/audio.py
new file mode 100644
index 0000000..67b568c
--- /dev/null
+++ b/llamachat/audio.py
@@ -0,0 +1,116 @@
+# SPDX-License-Identifier: GPL-2.0-only
+#
+# llamachat - a small native chat client for a local llama.cpp router
+# Copyright (C) 2026 Danilo M. <danix@danix.xyz>
+#
+# This program is free software; you can redistribute it and/or modify
+# it under the terms of the GNU General Public License version 2 as
+# published by the Free Software Foundation.
+#
+# This program is distributed in the hope that it will be useful,
+# but WITHOUT ANY WARRANTY; without even the implied warranty of
+# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
+# GNU General Public License for more details.
+"""Microphone capture for audio-capable models."""
+
+import io
+import wave
+
+try:
+ from PySide6.QtCore import (
+ QBuffer,
+ QByteArray,
+ QIODevice,
+ QObject,
+ QTimer,
+ Signal,
+ )
+ from PySide6.QtMultimedia import QAudioFormat, QAudioSource, QMediaDevices
+
+ AVAILABLE = True
+except ImportError:
+ # PySide6-Essentials ships without QtMultimedia. The feature is optional:
+ # the record control is hidden and nothing else changes.
+ AVAILABLE = False
+
+# Speech encoders expect 16 kHz mono. Int16 is the sample format the WAVE
+# header below declares, so SAMPLE_BYTES and the QString format must agree.
+RATE = 16000
+CHANNELS = 1
+SAMPLE_BYTES = 2
+MAX_SECONDS = 60
+MIN_SECONDS = 0.3
+
+
+def to_wav(pcm: bytes, rate: int = RATE, channels: int = CHANNELS) -> bytes:
+ """Wrap raw Int16 PCM in a RIFF/WAVE header.
+
+ No Qt in here on purpose: this is the one piece of the recorder that can
+ be tested without a sound device.
+ """
+ buf = io.BytesIO()
+ with wave.open(buf, "wb") as wav:
+ wav.setnchannels(channels)
+ wav.setsampwidth(SAMPLE_BYTES)
+ wav.setframerate(rate)
+ wav.writeframes(pcm)
+ return buf.getvalue()
+
+
+if AVAILABLE:
+
+ class AudioRecorder(QObject):
+ """Records the default input device to raw PCM at 16 kHz mono.
+
+ The UI drives it: start(), then stop() for the PCM. A clip that hits
+ MAX_SECONDS is halted by the recorder, which emits `capped` so the UI
+ can finalize it without waiting for a click.
+ """
+
+ capped = Signal()
+
+ def __init__(self, parent=None):
+ super().__init__(parent)
+ self._buffer = QByteArray()
+ self._io = None
+ self._source = None
+ self.timed_out = False
+ self._timer = QTimer(self)
+ self._timer.setSingleShot(True)
+ self._timer.timeout.connect(self._on_cap)
+
+ def start(self) -> bool:
+ """Begin capture. False when there is no input device."""
+ device = QMediaDevices.defaultAudioInput()
+ if device.isNull():
+ return False
+ fmt = QAudioFormat()
+ fmt.setSampleRate(RATE)
+ fmt.setChannelCount(CHANNELS)
+ fmt.setSampleFormat(QAudioFormat.Int16)
+ self._buffer = QByteArray()
+ self.timed_out = False
+ self._io = QBuffer(self._buffer)
+ self._io.open(QIODevice.WriteOnly)
+ self._source = QAudioSource(device, fmt, self)
+ self._source.start(self._io)
+ self._timer.start(MAX_SECONDS * 1000)
+ return True
+
+ def stop(self) -> bytes:
+ """Stop capture and return the raw PCM. Safe to call twice."""
+ self._halt()
+ return bytes(self._buffer)
+
+ def _halt(self) -> None:
+ if self._source is None:
+ return
+ self._timer.stop()
+ self._source.stop()
+ self._source = None
+ self._io.close()
+
+ def _on_cap(self) -> None:
+ self.timed_out = True
+ self._halt()
+ self.capped.emit()
diff --git a/test_llamachat.py b/test_llamachat.py
index 8fb7a0e..586be47 100755
--- a/test_llamachat.py
+++ b/test_llamachat.py
@@ -15,6 +15,7 @@
"""Self-checks for the non-GUI logic. Run: ./test_llamachat.py"""
import contextlib
+import io
import json
import os
import subprocess
@@ -378,6 +379,30 @@ def test_user_content():
print("ok user content assembly")
+def test_audio_wav():
+ """Recorded PCM becomes a readable WAVE file."""
+ import wave
+
+ from llamachat import audio
+
+ # Eight bytes: four Int16 samples.
+ pcm = b"\x01\x00\x02\x00\x03\x00\x04\x00"
+ wav = audio.to_wav(pcm)
+ assert wav[:4] == b"RIFF"
+ assert wav[8:12] == b"WAVE"
+
+ with wave.open(io.BytesIO(wav), "rb") as fh:
+ assert fh.getnchannels() == 1
+ assert fh.getsampwidth() == 2
+ assert fh.getframerate() == 16000
+ assert fh.getnframes() == 4
+ assert fh.readframes(4) == pcm
+
+ # The recorder is optional: AVAILABLE must exist as a bool either way.
+ assert isinstance(audio.AVAILABLE, bool)
+ print("ok audio wav encoding")
+
+
def test_markdown_rendering():
"""Replies render as markdown; markup inside them stays literal."""
import os
@@ -3847,6 +3872,7 @@ if __name__ == "__main__":
test_reasoning_storage()
test_migration_adds_reasoning()
test_user_content()
+ test_audio_wav()
test_prompt_store()
test_token_estimate()
test_usage_parsing()