From 0a0702f15db4d3f739049b5f0a9882ad1bcc631d Mon Sep 17 00:00:00 2001 From: avi Date: Wed, 16 Sep 2026 18:28:59 -0500 Subject: [PATCH] =?UTF-8?q?Summarize=20progress=20counts=20reasoning=20tok?= =?UTF-8?q?ens:=20thinking=20models=20stream=20most=20of=20the=20run=20in?= =?UTF-8?q?=20reasoning=5Fcontent,=20which=20the=20ticker=20ignored=20?= =?UTF-8?q?=E2=80=94=20UI=20sat=20at=20'waiting=20for=20model'=20then=20ju?= =?UTF-8?q?mped=20to=20done=20with=20no=20percentage?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- backend/shonar/services/ai/ollama.py | 8 ++++-- backend/shonar/services/ai/openai_compat.py | 11 ++++++-- backend/tests/test_ai_adapters.py | 28 +++++++++++++++++++++ 3 files changed, 43 insertions(+), 4 deletions(-) diff --git a/backend/shonar/services/ai/ollama.py b/backend/shonar/services/ai/ollama.py index 424f96a..7edf1ea 100644 --- a/backend/shonar/services/ai/ollama.py +++ b/backend/shonar/services/ai/ollama.py @@ -117,9 +117,13 @@ async def _collect_reply(resp: httpx.Response, ticker) -> str: except ValueError: continue piece = (obj.get("message") or {}).get("content") or "" - if piece: + # Like the openai_compat adapter: a thinking channel is real + # work and must move the progress ticker even when the server + # ignored think:false. + thinking = (obj.get("message") or {}).get("thinking") or "" + if piece or thinking: parts.append(piece) - total += len(piece) + total += len(piece) + len(thinking) ticker(total) if obj.get("done"): break diff --git a/backend/shonar/services/ai/openai_compat.py b/backend/shonar/services/ai/openai_compat.py index 0fe4d9a..49cb1bb 100644 --- a/backend/shonar/services/ai/openai_compat.py +++ b/backend/shonar/services/ai/openai_compat.py @@ -48,11 +48,18 @@ async def _collect_reply(resp: httpx.Response, ticker) -> str: chunk = _json.loads(data) delta = chunk["choices"][0].get("delta") or {} piece = delta.get("content") or "" + # Thinking models (qwen3 on llama.cpp) stream a long + # reasoning_content channel BEFORE any content: counting only + # content left the progress bar frozen at "waiting for model" + # for the entire run, then jumping straight to done. Reasoning + # is real work — count it toward progress (never into the + # reply itself). + thinking = delta.get("reasoning_content") or "" except (ValueError, KeyError, IndexError, TypeError): continue # keep-alives / usage chunks / odd frames: not content - if piece: + if piece or thinking: parts.append(piece) - total += len(piece) + total += len(piece) + len(thinking) ticker(total) content = "".join(parts) if not content: diff --git a/backend/tests/test_ai_adapters.py b/backend/tests/test_ai_adapters.py index 5c7d260..ff295c7 100644 --- a/backend/tests/test_ai_adapters.py +++ b/backend/tests/test_ai_adapters.py @@ -155,6 +155,34 @@ async def test_openai_compat_streams_with_progress(): assert all(0 <= p <= 99 for p in seen_pcts) # never claims 100 early +async def test_openai_compat_thinking_moves_progress_before_content(): + """qwen3-on-llama.cpp streams reasoning_content for most of the run; + progress must climb during that phase, not sit at "waiting for model" + and then jump straight to done.""" + import json as _json + + def sse(request: httpx.Request) -> httpx.Response: + think = [f"reasoning chunk {i} " for i in range(40)] # ~800 chars + lines = [f"data: {_json.dumps({'choices': [{'delta': {'reasoning_content': p}}]})}" + for p in think] + answer = _json.dumps({"short": "Done.", "key_points": []}) + lines.append(f"data: {_json.dumps({'choices': [{'delta': {'content': answer}}]})}") + lines.append("data: [DONE]") + return httpx.Response( + 200, content=("\n\n".join(lines) + "\n\n").encode(), + headers={"content-type": "text/event-stream"}) + + p = OpenAICompatProvider("http://llm:8000", model="qwen", + http_client=mock_client(sse)) + seen_pcts = [] + res = await p.summarize("t", on_progress=seen_pcts.append) + assert res.short == "Done." # reasoning never enters the reply + # The ticker must report real mid-run values while only reasoning + # streamed — otherwise the UI shows no percentage for the whole job. + assert len(seen_pcts) > 10 + assert seen_pcts[-1] > 50 + + async def test_openai_compat_partial_json_gets_defaults(): import json as _json