From 99b849463b6748dbb14f4d4cdfa18185e2a860a9 Mon Sep 17 00:00:00 2001 From: "Danilo M." Date: Fri, 31 Jul 2026 11:40:34 +0200 Subject: fix: keep reasoning out of the reply body A delta carrying both reasoning_content and an empty content was classified by truthiness, so the reasoning fell through to the content branch. The tail of the model's thinking was stored and displayed as the answer, splitting mid-sentence across the two fields. Classify on the presence of the field instead, and return nothing for an empty reasoning delta rather than falling through. The visible symptom was a code block swallowing the message: leaked thinking is dense with backticks, and an odd count leaves a run open to the end of the reply. Close an unterminated fence before parsing, with a closer matching the opener's length, and drop a lone dangling inline backtick. Streaming needs this regardless, since a fence is unclosed on nearly every frame. Co-Authored-By: Claude Opus 5 --- llamachat/backend.py | 7 +++++-- llamachat/ui.py | 29 ++++++++++++++++++++++++++++- test_llamachat.py | 51 +++++++++++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 84 insertions(+), 3 deletions(-) diff --git a/llamachat/backend.py b/llamachat/backend.py index 9e0556f..ec3b7d6 100644 --- a/llamachat/backend.py +++ b/llamachat/backend.py @@ -267,9 +267,12 @@ def _parse_sse_line(line: str) -> tuple[str, str] | None: return None delta = choices[0].get("delta") or {} + # Both fields can arrive in one delta, and a reasoning delta can be an + # empty string. Test for presence, not truthiness, so a chunk carrying + # reasoning is never mistaken for a content chunk. reasoning = delta.get("reasoning_content") - if reasoning: - return ("reasoning", reasoning) + if reasoning is not None: + return ("reasoning", reasoning) if reasoning else None content = delta.get("content") if content: return ("content", content) diff --git a/llamachat/ui.py b/llamachat/ui.py index fd2380c..f599bdd 100644 --- a/llamachat/ui.py +++ b/llamachat/ui.py @@ -1291,6 +1291,33 @@ _PRE_STYLE = ( ) +_FENCE_RE = re.compile(r"^\s{0,3}(`{3,}|~{3,})", re.MULTILINE) + + +def _close_fences(text: str) -> str: + """Close an unterminated code fence and neuter a lone inline backtick. + + A reply is streamed, so a fence is unclosed on nearly every frame; a + model can also emit an odd backtick by accident. Either way the rest of + the reply would reflow as code, which is how a whole message turns into + one grey block. Closing the fence here bounds the damage to the text + that was really meant as code. + """ + fences = _FENCE_RE.findall(text) + if len(fences) % 2: + # Match the opener's own length; a longer run is legal and only a + # fence at least as long as the opener actually closes it. + closer = fences[-1][0] * max(3, len(fences[-1])) + nl = "" if text.endswith("\n") else "\n" + return f"{text}{nl}{closer}" + if not fences and text.count("`") % 2: + # No fences at all, one dangling inline tick: drop it rather than + # let it open a run to the end of the message. + head, _, tail = text.rpartition("`") + return head + tail + return text + + def _markdown_to_fragment(text: str) -> str: """Render markdown to an HTML fragment safe to splice into the page. @@ -1300,7 +1327,7 @@ def _markdown_to_fragment(text: str) -> str: if not text: return "" doc = QTextDocument() - doc.setMarkdown(text, MARKDOWN_FLAGS) + doc.setMarkdown(_close_fences(text), MARKDOWN_FLAGS) html_text = doc.toHtml() start = html_text.find("