From dae64bcaf9c9a8a8eda134297e5844efc1f4c9c0 Mon Sep 17 00:00:00 2001 From: "Danilo M." Date: Fri, 18 Sep 2026 20:21:04 +0200 Subject: feat: audio attachment kind and input_audio content part --- test_llamachat.py | 48 ++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 48 insertions(+) (limited to 'test_llamachat.py') diff --git a/test_llamachat.py b/test_llamachat.py index 586be47..44f569e 100755 --- a/test_llamachat.py +++ b/test_llamachat.py @@ -229,12 +229,28 @@ def test_attachment_truncation(): print("ok attachment truncation") +def test_audio_load(): + """A .wav file loads as an audio attachment, not as text or unknown.""" + with tempfile.TemporaryDirectory() as tmp: + src = Path(tmp) / "note.wav" + src.write_bytes(b"RIFF\x00\x00\x00\x00WAVEfmt ") + att = backend.load_attachment(src, 1000) + assert att.kind == "audio" + assert att.b64 + assert not att.b64.startswith("data:") + assert att.thumb is None + assert att.text == "" + print("ok audio attachment load") + + def test_classify(): assert backend.classify(Path("a.py")) == "text" assert backend.classify(Path("a.SlackBuild")) == "text" assert backend.classify(Path("a.png")) == "image" assert backend.classify(Path("a.jpg")) == "image" assert backend.classify(Path("a.so")) == "unknown" + assert backend.classify(Path("a.wav")) == "audio" + assert backend.classify(Path("a.WAV")) == "audio" print("ok file classification") @@ -376,6 +392,25 @@ def test_user_content(): assert content[0]["type"] == "text" assert content[1]["type"] == "image_url" assert content[1]["image_url"]["url"].startswith("data:image/png") + + # Audio -> the input_audio part llama.cpp expects, raw base64. The + # data must not carry a data: prefix, unlike the image URL above. + rec = backend.audio_attachment(b"\x00\x01" * 8, seconds=2) + assert rec.kind == "audio" + assert rec.path.name == "voice-note-2s.wav" + assert rec.sha256 and rec.size == 16 + + content = backend.build_user_content("listen", [rec]) + assert isinstance(content, list) + assert content[0] == {"type": "text", "text": "listen"} + assert content[1]["type"] == "input_audio" + assert content[1]["input_audio"]["format"] == "wav" + assert content[1]["input_audio"]["data"] == rec.b64 + assert not content[1]["input_audio"]["data"].startswith("data:") + + # Image and audio together keep both parts, text first. + both = backend.build_user_content("look and listen", [img, rec]) + assert [p["type"] for p in both] == ["text", "image_url", "input_audio"] print("ok user content assembly") @@ -758,6 +793,18 @@ def test_token_estimate(): ] assert backend.estimate_tokens(with_image, 3.5) > 500 + # An audio part costs like an image, not free. + with_audio = [ + { + "role": "user", + "content": [ + {"type": "text", "text": "listen"}, + {"type": "input_audio", "input_audio": {"data": "AAA", "format": "wav"}}, + ], + } + ] + assert backend.estimate_tokens(with_audio, 3.5) > 500 + # A silly ratio must not divide by zero. assert backend.estimate_tokens(small, 0) > 0 print("ok token estimate") @@ -3867,6 +3914,7 @@ if __name__ == "__main__": test_fts_query_escaping() test_history_roundtrip() test_attachment_truncation() + test_audio_load() test_classify() test_sse_parsing() test_reasoning_storage() -- cgit v1.2.3