Summarize streams with real progress: adapters reassemble SSE/NDJSON deltas and report 0-99% as tokens arrive

This commit is contained in:
avi 2026-09-15 19:45:02 -05:00
commit 7fa99e8291
8 changed files with 246 additions and 42 deletions

View file

@ -8,7 +8,12 @@ from contextlib import asynccontextmanager
import httpx
from shonar.services.ai import ProviderConfigError, ProviderTransientError, SummaryResult
from shonar.services.ai._llm import SYSTEM_PROMPT, build_user_message, parse_summary
from shonar.services.ai._llm import (
SYSTEM_PROMPT,
build_user_message,
make_progress_ticker,
parse_summary,
)
class OllamaProvider:
@ -39,10 +44,13 @@ class OllamaProvider:
yield client
async def summarize(self, transcript: str, *, title: str | None = None,
tone: str | None = None) -> SummaryResult:
tone: str | None = None, on_progress=None) -> SummaryResult:
ticker = make_progress_ticker(on_progress)
payload = {
"model": self.model,
"stream": False,
# Stream (NDJSON lines) so the UI shows real progress while the
# model generates; the reply is reassembled from the chunks.
"stream": True,
"format": "json",
# qwen3-family models "think" by default: a long reasoning
# chain before the JSON answer, brutally slow on CPU and it
@ -57,20 +65,20 @@ class OllamaProvider:
],
}
try:
async with self._client() as client:
resp = await client.post(f"{self.base_url}/api/chat", json=payload)
async with self._client() as client, client.stream(
"POST", f"{self.base_url}/api/chat", json=payload
) as resp:
if resp.status_code == 404:
# Missing model and missing route both 404 here; both
# are configuration, not weather.
raise ProviderConfigError(
"Ollama has no such model or route (HTTP 404).")
if resp.status_code != 200:
raise ProviderTransientError(
f"Summarization failed (HTTP {resp.status_code}).")
content = await _collect_reply(resp, ticker)
except (httpx.TimeoutException, httpx.TransportError) as e:
raise ProviderTransientError(f"Ollama unreachable: {type(e).__name__}") from e
if resp.status_code == 404:
# Missing model and missing route both 404 here; both are
# configuration, not weather.
raise ProviderConfigError("Ollama has no such model or route (HTTP 404).")
if resp.status_code != 200:
raise ProviderTransientError(f"Summarization failed (HTTP {resp.status_code}).")
try:
content = resp.json()["message"]["content"]
except (ValueError, KeyError, TypeError) as e:
raise ProviderTransientError("Ollama sent an unreadable reply.") from e
import json as _json
try:
@ -78,3 +86,44 @@ class OllamaProvider:
except ValueError as e:
raise ProviderTransientError("Ollama reply was not JSON.") from e
return parse_summary(data, self.model)
async def _collect_reply(resp: httpx.Response, ticker) -> str:
"""Reassemble the assistant reply from Ollama's streamed chat response.
Streaming answers are NDJSON (one JSON object per line, ``done`` on the
last); a server honoring stream=false returns one JSON body — both work.
The running character count feeds [ticker] for UI progress."""
import json as _json
ctype = resp.headers.get("content-type", "")
if "ndjson" not in ctype and "event-stream" not in ctype:
body = await resp.aread()
try:
content = _json.loads(body)["message"]["content"]
except (ValueError, KeyError, TypeError) as e:
raise ProviderTransientError("Ollama sent an unreadable reply.") from e
ticker(len(content or ""))
return content or ""
parts: list[str] = []
total = 0
async for line in resp.aiter_lines():
line = line.strip()
if not line:
continue
try:
obj = _json.loads(line)
except ValueError:
continue
piece = (obj.get("message") or {}).get("content") or ""
if piece:
parts.append(piece)
total += len(piece)
ticker(total)
if obj.get("done"):
break
content = "".join(parts)
if not content:
raise ProviderTransientError("Ollama sent an unreadable reply.")
return content