Summarize progress counts reasoning tokens: thinking models stream most of the run in reasoning_content, which the ticker ignored — UI sat at 'waiting for model' then jumped to done with no percentage
This commit is contained in:
parent
8f7e1f9ea5
commit
0a0702f15d
3 changed files with 43 additions and 4 deletions
|
|
@ -117,9 +117,13 @@ async def _collect_reply(resp: httpx.Response, ticker) -> str:
|
||||||
except ValueError:
|
except ValueError:
|
||||||
continue
|
continue
|
||||||
piece = (obj.get("message") or {}).get("content") or ""
|
piece = (obj.get("message") or {}).get("content") or ""
|
||||||
if piece:
|
# Like the openai_compat adapter: a thinking channel is real
|
||||||
|
# work and must move the progress ticker even when the server
|
||||||
|
# ignored think:false.
|
||||||
|
thinking = (obj.get("message") or {}).get("thinking") or ""
|
||||||
|
if piece or thinking:
|
||||||
parts.append(piece)
|
parts.append(piece)
|
||||||
total += len(piece)
|
total += len(piece) + len(thinking)
|
||||||
ticker(total)
|
ticker(total)
|
||||||
if obj.get("done"):
|
if obj.get("done"):
|
||||||
break
|
break
|
||||||
|
|
|
||||||
|
|
@ -48,11 +48,18 @@ async def _collect_reply(resp: httpx.Response, ticker) -> str:
|
||||||
chunk = _json.loads(data)
|
chunk = _json.loads(data)
|
||||||
delta = chunk["choices"][0].get("delta") or {}
|
delta = chunk["choices"][0].get("delta") or {}
|
||||||
piece = delta.get("content") or ""
|
piece = delta.get("content") or ""
|
||||||
|
# Thinking models (qwen3 on llama.cpp) stream a long
|
||||||
|
# reasoning_content channel BEFORE any content: counting only
|
||||||
|
# content left the progress bar frozen at "waiting for model"
|
||||||
|
# for the entire run, then jumping straight to done. Reasoning
|
||||||
|
# is real work — count it toward progress (never into the
|
||||||
|
# reply itself).
|
||||||
|
thinking = delta.get("reasoning_content") or ""
|
||||||
except (ValueError, KeyError, IndexError, TypeError):
|
except (ValueError, KeyError, IndexError, TypeError):
|
||||||
continue # keep-alives / usage chunks / odd frames: not content
|
continue # keep-alives / usage chunks / odd frames: not content
|
||||||
if piece:
|
if piece or thinking:
|
||||||
parts.append(piece)
|
parts.append(piece)
|
||||||
total += len(piece)
|
total += len(piece) + len(thinking)
|
||||||
ticker(total)
|
ticker(total)
|
||||||
content = "".join(parts)
|
content = "".join(parts)
|
||||||
if not content:
|
if not content:
|
||||||
|
|
|
||||||
|
|
@ -155,6 +155,34 @@ async def test_openai_compat_streams_with_progress():
|
||||||
assert all(0 <= p <= 99 for p in seen_pcts) # never claims 100 early
|
assert all(0 <= p <= 99 for p in seen_pcts) # never claims 100 early
|
||||||
|
|
||||||
|
|
||||||
|
async def test_openai_compat_thinking_moves_progress_before_content():
|
||||||
|
"""qwen3-on-llama.cpp streams reasoning_content for most of the run;
|
||||||
|
progress must climb during that phase, not sit at "waiting for model"
|
||||||
|
and then jump straight to done."""
|
||||||
|
import json as _json
|
||||||
|
|
||||||
|
def sse(request: httpx.Request) -> httpx.Response:
|
||||||
|
think = [f"reasoning chunk {i} " for i in range(40)] # ~800 chars
|
||||||
|
lines = [f"data: {_json.dumps({'choices': [{'delta': {'reasoning_content': p}}]})}"
|
||||||
|
for p in think]
|
||||||
|
answer = _json.dumps({"short": "Done.", "key_points": []})
|
||||||
|
lines.append(f"data: {_json.dumps({'choices': [{'delta': {'content': answer}}]})}")
|
||||||
|
lines.append("data: [DONE]")
|
||||||
|
return httpx.Response(
|
||||||
|
200, content=("\n\n".join(lines) + "\n\n").encode(),
|
||||||
|
headers={"content-type": "text/event-stream"})
|
||||||
|
|
||||||
|
p = OpenAICompatProvider("http://llm:8000", model="qwen",
|
||||||
|
http_client=mock_client(sse))
|
||||||
|
seen_pcts = []
|
||||||
|
res = await p.summarize("t", on_progress=seen_pcts.append)
|
||||||
|
assert res.short == "Done." # reasoning never enters the reply
|
||||||
|
# The ticker must report real mid-run values while only reasoning
|
||||||
|
# streamed — otherwise the UI shows no percentage for the whole job.
|
||||||
|
assert len(seen_pcts) > 10
|
||||||
|
assert seen_pcts[-1] > 50
|
||||||
|
|
||||||
|
|
||||||
async def test_openai_compat_partial_json_gets_defaults():
|
async def test_openai_compat_partial_json_gets_defaults():
|
||||||
import json as _json
|
import json as _json
|
||||||
|
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue