diff --git a/backend/shonar/services/ai/ollama.py b/backend/shonar/services/ai/ollama.py index 424f96a..7edf1ea 100644 --- a/backend/shonar/services/ai/ollama.py +++ b/backend/shonar/services/ai/ollama.py @@ -117,9 +117,13 @@ async def _collect_reply(resp: httpx.Response, ticker) -> str: except ValueError: continue piece = (obj.get("message") or {}).get("content") or "" - if piece: + # Like the openai_compat adapter: a thinking channel is real + # work and must move the progress ticker even when the server + # ignored think:false. + thinking = (obj.get("message") or {}).get("thinking") or "" + if piece or thinking: parts.append(piece) - total += len(piece) + total += len(piece) + len(thinking) ticker(total) if obj.get("done"): break diff --git a/backend/shonar/services/ai/openai_compat.py b/backend/shonar/services/ai/openai_compat.py index 0fe4d9a..49cb1bb 100644 --- a/backend/shonar/services/ai/openai_compat.py +++ b/backend/shonar/services/ai/openai_compat.py @@ -48,11 +48,18 @@ async def _collect_reply(resp: httpx.Response, ticker) -> str: chunk = _json.loads(data) delta = chunk["choices"][0].get("delta") or {} piece = delta.get("content") or "" + # Thinking models (qwen3 on llama.cpp) stream a long + # reasoning_content channel BEFORE any content: counting only + # content left the progress bar frozen at "waiting for model" + # for the entire run, then jumping straight to done. Reasoning + # is real work — count it toward progress (never into the + # reply itself). + thinking = delta.get("reasoning_content") or "" except (ValueError, KeyError, IndexError, TypeError): continue # keep-alives / usage chunks / odd frames: not content - if piece: + if piece or thinking: parts.append(piece) - total += len(piece) + total += len(piece) + len(thinking) ticker(total) content = "".join(parts) if not content: diff --git a/backend/tests/test_ai_adapters.py b/backend/tests/test_ai_adapters.py index 5c7d260..ff295c7 100644 --- a/backend/tests/test_ai_adapters.py +++ b/backend/tests/test_ai_adapters.py @@ -155,6 +155,34 @@ async def test_openai_compat_streams_with_progress(): assert all(0 <= p <= 99 for p in seen_pcts) # never claims 100 early +async def test_openai_compat_thinking_moves_progress_before_content(): + """qwen3-on-llama.cpp streams reasoning_content for most of the run; + progress must climb during that phase, not sit at "waiting for model" + and then jump straight to done.""" + import json as _json + + def sse(request: httpx.Request) -> httpx.Response: + think = [f"reasoning chunk {i} " for i in range(40)] # ~800 chars + lines = [f"data: {_json.dumps({'choices': [{'delta': {'reasoning_content': p}}]})}" + for p in think] + answer = _json.dumps({"short": "Done.", "key_points": []}) + lines.append(f"data: {_json.dumps({'choices': [{'delta': {'content': answer}}]})}") + lines.append("data: [DONE]") + return httpx.Response( + 200, content=("\n\n".join(lines) + "\n\n").encode(), + headers={"content-type": "text/event-stream"}) + + p = OpenAICompatProvider("http://llm:8000", model="qwen", + http_client=mock_client(sse)) + seen_pcts = [] + res = await p.summarize("t", on_progress=seen_pcts.append) + assert res.short == "Done." # reasoning never enters the reply + # The ticker must report real mid-run values while only reasoning + # streamed — otherwise the UI shows no percentage for the whole job. + assert len(seen_pcts) > 10 + assert seen_pcts[-1] > 50 + + async def test_openai_compat_partial_json_gets_defaults(): import json as _json