diff options
| -rw-r--r-- | llamachat/audio.py | 116 | ||||
| -rwxr-xr-x | test_llamachat.py | 26 |
2 files changed, 142 insertions, 0 deletions
diff --git a/llamachat/audio.py b/llamachat/audio.py new file mode 100644 index 0000000..67b568c --- /dev/null +++ b/llamachat/audio.py @@ -0,0 +1,116 @@ +# SPDX-License-Identifier: GPL-2.0-only +# +# llamachat - a small native chat client for a local llama.cpp router +# Copyright (C) 2026 Danilo M. <danix@danix.xyz> +# +# This program is free software; you can redistribute it and/or modify +# it under the terms of the GNU General Public License version 2 as +# published by the Free Software Foundation. +# +# This program is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +"""Microphone capture for audio-capable models.""" + +import io +import wave + +try: + from PySide6.QtCore import ( + QBuffer, + QByteArray, + QIODevice, + QObject, + QTimer, + Signal, + ) + from PySide6.QtMultimedia import QAudioFormat, QAudioSource, QMediaDevices + + AVAILABLE = True +except ImportError: + # PySide6-Essentials ships without QtMultimedia. The feature is optional: + # the record control is hidden and nothing else changes. + AVAILABLE = False + +# Speech encoders expect 16 kHz mono. Int16 is the sample format the WAVE +# header below declares, so SAMPLE_BYTES and the QString format must agree. +RATE = 16000 +CHANNELS = 1 +SAMPLE_BYTES = 2 +MAX_SECONDS = 60 +MIN_SECONDS = 0.3 + + +def to_wav(pcm: bytes, rate: int = RATE, channels: int = CHANNELS) -> bytes: + """Wrap raw Int16 PCM in a RIFF/WAVE header. + + No Qt in here on purpose: this is the one piece of the recorder that can + be tested without a sound device. + """ + buf = io.BytesIO() + with wave.open(buf, "wb") as wav: + wav.setnchannels(channels) + wav.setsampwidth(SAMPLE_BYTES) + wav.setframerate(rate) + wav.writeframes(pcm) + return buf.getvalue() + + +if AVAILABLE: + + class AudioRecorder(QObject): + """Records the default input device to raw PCM at 16 kHz mono. + + The UI drives it: start(), then stop() for the PCM. A clip that hits + MAX_SECONDS is halted by the recorder, which emits `capped` so the UI + can finalize it without waiting for a click. + """ + + capped = Signal() + + def __init__(self, parent=None): + super().__init__(parent) + self._buffer = QByteArray() + self._io = None + self._source = None + self.timed_out = False + self._timer = QTimer(self) + self._timer.setSingleShot(True) + self._timer.timeout.connect(self._on_cap) + + def start(self) -> bool: + """Begin capture. False when there is no input device.""" + device = QMediaDevices.defaultAudioInput() + if device.isNull(): + return False + fmt = QAudioFormat() + fmt.setSampleRate(RATE) + fmt.setChannelCount(CHANNELS) + fmt.setSampleFormat(QAudioFormat.Int16) + self._buffer = QByteArray() + self.timed_out = False + self._io = QBuffer(self._buffer) + self._io.open(QIODevice.WriteOnly) + self._source = QAudioSource(device, fmt, self) + self._source.start(self._io) + self._timer.start(MAX_SECONDS * 1000) + return True + + def stop(self) -> bytes: + """Stop capture and return the raw PCM. Safe to call twice.""" + self._halt() + return bytes(self._buffer) + + def _halt(self) -> None: + if self._source is None: + return + self._timer.stop() + self._source.stop() + self._source = None + self._io.close() + + def _on_cap(self) -> None: + self.timed_out = True + self._halt() + self.capped.emit() diff --git a/test_llamachat.py b/test_llamachat.py index 8fb7a0e..586be47 100755 --- a/test_llamachat.py +++ b/test_llamachat.py @@ -15,6 +15,7 @@ """Self-checks for the non-GUI logic. Run: ./test_llamachat.py""" import contextlib +import io import json import os import subprocess @@ -378,6 +379,30 @@ def test_user_content(): print("ok user content assembly") +def test_audio_wav(): + """Recorded PCM becomes a readable WAVE file.""" + import wave + + from llamachat import audio + + # Eight bytes: four Int16 samples. + pcm = b"\x01\x00\x02\x00\x03\x00\x04\x00" + wav = audio.to_wav(pcm) + assert wav[:4] == b"RIFF" + assert wav[8:12] == b"WAVE" + + with wave.open(io.BytesIO(wav), "rb") as fh: + assert fh.getnchannels() == 1 + assert fh.getsampwidth() == 2 + assert fh.getframerate() == 16000 + assert fh.getnframes() == 4 + assert fh.readframes(4) == pcm + + # The recorder is optional: AVAILABLE must exist as a bool either way. + assert isinstance(audio.AVAILABLE, bool) + print("ok audio wav encoding") + + def test_markdown_rendering(): """Replies render as markdown; markup inside them stays literal.""" import os @@ -3847,6 +3872,7 @@ if __name__ == "__main__": test_reasoning_storage() test_migration_adds_reasoning() test_user_content() + test_audio_wav() test_prompt_store() test_token_estimate() test_usage_parsing() |
