Summarize progress counts reasoning tokens: thinking models stream most of the run in reasoning_content, which the ticker ignored — UI sat at 'waiting for model' then jumped to done with no percentage
This commit is contained in:
parent
8f7e1f9ea5
commit
0a0702f15d
3 changed files with 43 additions and 4 deletions
|
|
@ -117,9 +117,13 @@ async def _collect_reply(resp: httpx.Response, ticker) -> str:
|
|||
except ValueError:
|
||||
continue
|
||||
piece = (obj.get("message") or {}).get("content") or ""
|
||||
if piece:
|
||||
# Like the openai_compat adapter: a thinking channel is real
|
||||
# work and must move the progress ticker even when the server
|
||||
# ignored think:false.
|
||||
thinking = (obj.get("message") or {}).get("thinking") or ""
|
||||
if piece or thinking:
|
||||
parts.append(piece)
|
||||
total += len(piece)
|
||||
total += len(piece) + len(thinking)
|
||||
ticker(total)
|
||||
if obj.get("done"):
|
||||
break
|
||||
|
|
|
|||
|
|
@ -48,11 +48,18 @@ async def _collect_reply(resp: httpx.Response, ticker) -> str:
|
|||
chunk = _json.loads(data)
|
||||
delta = chunk["choices"][0].get("delta") or {}
|
||||
piece = delta.get("content") or ""
|
||||
# Thinking models (qwen3 on llama.cpp) stream a long
|
||||
# reasoning_content channel BEFORE any content: counting only
|
||||
# content left the progress bar frozen at "waiting for model"
|
||||
# for the entire run, then jumping straight to done. Reasoning
|
||||
# is real work — count it toward progress (never into the
|
||||
# reply itself).
|
||||
thinking = delta.get("reasoning_content") or ""
|
||||
except (ValueError, KeyError, IndexError, TypeError):
|
||||
continue # keep-alives / usage chunks / odd frames: not content
|
||||
if piece:
|
||||
if piece or thinking:
|
||||
parts.append(piece)
|
||||
total += len(piece)
|
||||
total += len(piece) + len(thinking)
|
||||
ticker(total)
|
||||
content = "".join(parts)
|
||||
if not content:
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue