Summarize streams with real progress: adapters reassemble SSE/NDJSON deltas and report 0-99% as tokens arrive

This commit is contained in:
avi 2026-09-15 19:45:02 -05:00
commit 7fa99e8291
8 changed files with 246 additions and 42 deletions

View file

@ -113,7 +113,8 @@ class LlmProvider(Protocol):
name: str
async def summarize(self, transcript: str, *, title: str | None = None,
tone: str | None = None) -> SummaryResult: ...
tone: str | None = None,
on_progress=None) -> SummaryResult: ... # Callable[[int], None] | None
def get_transcription_provider(settings: Settings) -> TranscriptionProvider | None:

View file

@ -38,6 +38,30 @@ def build_user_message(transcript: str, title: str | None,
return msg
# Rough typical length of a summary JSON reply. The stream cannot know the
# model's final size, so progress is an estimate that crawls toward 99 and
# the job sets 100 on success — honest "almost there", never a fake jump.
EST_OUTPUT_CHARS = 1200
def make_progress_ticker(on_progress):
"""Wrap an optional on_progress callback into a feed(n_chars) sink.
Reports a clamped 0..99 estimate from streamed output size. Silent
no-op when the caller has no callback; failures never disturb the
summary itself."""
if on_progress is None:
return lambda n_chars: None
def feed(n_chars: int) -> None:
try: # noqa: SIM105 — swallow deliberately: progress is display state
on_progress(min(99, n_chars * 100 // EST_OUTPUT_CHARS))
except Exception: # display state only
pass
return feed
def parse_summary(data: object, model: str) -> SummaryResult:
if not isinstance(data, dict):
return SummaryResult(model=model)

View file

@ -8,7 +8,12 @@ from contextlib import asynccontextmanager
import httpx
from shonar.services.ai import ProviderConfigError, ProviderTransientError, SummaryResult
from shonar.services.ai._llm import SYSTEM_PROMPT, build_user_message, parse_summary
from shonar.services.ai._llm import (
SYSTEM_PROMPT,
build_user_message,
make_progress_ticker,
parse_summary,
)
class OllamaProvider:
@ -39,10 +44,13 @@ class OllamaProvider:
yield client
async def summarize(self, transcript: str, *, title: str | None = None,
tone: str | None = None) -> SummaryResult:
tone: str | None = None, on_progress=None) -> SummaryResult:
ticker = make_progress_ticker(on_progress)
payload = {
"model": self.model,
"stream": False,
# Stream (NDJSON lines) so the UI shows real progress while the
# model generates; the reply is reassembled from the chunks.
"stream": True,
"format": "json",
# qwen3-family models "think" by default: a long reasoning
# chain before the JSON answer, brutally slow on CPU and it
@ -57,20 +65,20 @@ class OllamaProvider:
],
}
try:
async with self._client() as client:
resp = await client.post(f"{self.base_url}/api/chat", json=payload)
async with self._client() as client, client.stream(
"POST", f"{self.base_url}/api/chat", json=payload
) as resp:
if resp.status_code == 404:
# Missing model and missing route both 404 here; both
# are configuration, not weather.
raise ProviderConfigError(
"Ollama has no such model or route (HTTP 404).")
if resp.status_code != 200:
raise ProviderTransientError(
f"Summarization failed (HTTP {resp.status_code}).")
content = await _collect_reply(resp, ticker)
except (httpx.TimeoutException, httpx.TransportError) as e:
raise ProviderTransientError(f"Ollama unreachable: {type(e).__name__}") from e
if resp.status_code == 404:
# Missing model and missing route both 404 here; both are
# configuration, not weather.
raise ProviderConfigError("Ollama has no such model or route (HTTP 404).")
if resp.status_code != 200:
raise ProviderTransientError(f"Summarization failed (HTTP {resp.status_code}).")
try:
content = resp.json()["message"]["content"]
except (ValueError, KeyError, TypeError) as e:
raise ProviderTransientError("Ollama sent an unreadable reply.") from e
import json as _json
try:
@ -78,3 +86,44 @@ class OllamaProvider:
except ValueError as e:
raise ProviderTransientError("Ollama reply was not JSON.") from e
return parse_summary(data, self.model)
async def _collect_reply(resp: httpx.Response, ticker) -> str:
"""Reassemble the assistant reply from Ollama's streamed chat response.
Streaming answers are NDJSON (one JSON object per line, ``done`` on the
last); a server honoring stream=false returns one JSON body — both work.
The running character count feeds [ticker] for UI progress."""
import json as _json
ctype = resp.headers.get("content-type", "")
if "ndjson" not in ctype and "event-stream" not in ctype:
body = await resp.aread()
try:
content = _json.loads(body)["message"]["content"]
except (ValueError, KeyError, TypeError) as e:
raise ProviderTransientError("Ollama sent an unreadable reply.") from e
ticker(len(content or ""))
return content or ""
parts: list[str] = []
total = 0
async for line in resp.aiter_lines():
line = line.strip()
if not line:
continue
try:
obj = _json.loads(line)
except ValueError:
continue
piece = (obj.get("message") or {}).get("content") or ""
if piece:
parts.append(piece)
total += len(piece)
ticker(total)
if obj.get("done"):
break
content = "".join(parts)
if not content:
raise ProviderTransientError("Ollama sent an unreadable reply.")
return content

View file

@ -10,7 +10,54 @@ from contextlib import asynccontextmanager
import httpx
from shonar.services.ai import ProviderConfigError, ProviderTransientError, SummaryResult
from shonar.services.ai._llm import SYSTEM_PROMPT, build_user_message, parse_summary
from shonar.services.ai._llm import (
SYSTEM_PROMPT,
build_user_message,
make_progress_ticker,
parse_summary,
)
async def _collect_reply(resp: httpx.Response, ticker) -> str:
"""Reassemble the assistant reply from a (possibly streamed) response.
Feeds the running character count to [ticker] as deltas arrive so the
UI can show progress. Servers that ignored "stream": true answer with
a plain JSON body — that path is handled too."""
import json as _json
ctype = resp.headers.get("content-type", "")
if "text/event-stream" not in ctype:
body = await resp.aread()
try:
content = _json.loads(body)["choices"][0]["message"]["content"]
except (ValueError, KeyError, IndexError, TypeError) as e:
raise ProviderTransientError("LLM sent an unreadable reply.") from e
ticker(len(content or ""))
return content or ""
parts: list[str] = []
total = 0
async for line in resp.aiter_lines():
if not line.startswith("data:"):
continue
data = line[5:].strip()
if data == "[DONE]":
break
try:
chunk = _json.loads(data)
delta = chunk["choices"][0].get("delta") or {}
piece = delta.get("content") or ""
except (ValueError, KeyError, IndexError, TypeError):
continue # keep-alives / usage chunks / odd frames: not content
if piece:
parts.append(piece)
total += len(piece)
ticker(total)
content = "".join(parts)
if not content:
raise ProviderTransientError("LLM sent an unreadable reply.")
return content
class OpenAICompatProvider:
@ -43,7 +90,8 @@ class OpenAICompatProvider:
yield client
async def summarize(self, transcript: str, *, title: str | None = None,
tone: str | None = None) -> SummaryResult:
tone: str | None = None, on_progress=None) -> SummaryResult:
ticker = make_progress_ticker(on_progress)
headers = (
{"Authorization": f"Bearer {self.api_key}"} if self.api_key else {}
)
@ -51,6 +99,10 @@ class OpenAICompatProvider:
"model": self.model,
"temperature": 0.2,
"response_format": {"type": "json_object"},
# Stream so the UI gets real progress while the model works;
# the full reply is reassembled from the deltas. Servers that
# ignore "stream" still work — the non-SSE body is handled too.
"stream": True,
"messages": [
{"role": "system", "content": SYSTEM_PROMPT},
{"role": "user",
@ -58,23 +110,22 @@ class OpenAICompatProvider:
],
}
try:
async with self._client() as client:
resp = await client.post(
f"{self.base_url}/v1/chat/completions", headers=headers, json=payload
)
async with self._client() as client, client.stream(
"POST", f"{self.base_url}/v1/chat/completions",
headers=headers, json=payload,
) as resp:
if resp.status_code in (401, 403, 404):
raise ProviderConfigError(
f"LLM refused the request (HTTP {resp.status_code}).")
if resp.status_code == 429 or resp.status_code >= 500:
raise ProviderTransientError(
f"LLM busy (HTTP {resp.status_code}).")
if resp.status_code != 200:
raise ProviderTransientError(
f"Summarization failed (HTTP {resp.status_code}).")
content = await _collect_reply(resp, ticker)
except (httpx.TimeoutException, httpx.TransportError) as e:
raise ProviderTransientError(f"LLM unreachable: {type(e).__name__}") from e
if resp.status_code in (401, 403, 404):
raise ProviderConfigError(f"LLM refused the request (HTTP {resp.status_code}).")
if resp.status_code == 429 or resp.status_code >= 500:
raise ProviderTransientError(f"LLM busy (HTTP {resp.status_code}).")
if resp.status_code != 200:
raise ProviderTransientError(f"Summarization failed (HTTP {resp.status_code}).")
try:
body = resp.json()
content = body["choices"][0]["message"]["content"]
except (ValueError, KeyError, IndexError, TypeError) as e:
raise ProviderTransientError("LLM sent an unreadable reply.") from e
import json as _json
try: