S.H.O.N.A.R._Desktop_Companion/backend/shonar/services/ai/_llm.py
avi 39aaa04a00 Honest summarize progress from real completed work units
Backend: new ContractProgress parses the streamed summary JSON and
advances only when contract units finish — first token 5%, each closed
key 5→90 (arrays step per closed item), root close 95, stored=100.
Replaces the char-count ticker that counted reasoning chars against a
guessed 1200-char output and sat frozen at 99 for minutes. No timers,
no elapsed-time estimates; the unfinishable thinking span honestly
earns only the alive-tick. Both adapters (openai_compat, ollama) feed
the accumulated content prefix. /jobs now carries the job's tone.

App: summarize progress shows a real percentage + determinate bar with
the tone named ('Summarizing dry wit… 47% · 1:12') in both detail
widgets; indeterminate only while queued or pre-first-token.

Tests: milestone sequence verified identical for char-by-char and
chunked streaming; adapter thinking-phase test updated to the new
contract (83 passed).
2026-09-18 15:42:16 -05:00

214 lines
9 KiB
Python

"""Shared summary contract: one system prompt, one JSON shape.
The model replies with JSON only; partial replies are accepted and missing
keys default to empty (a terse-but-valid summary beats a failed job).
Transcript input is truncated to bound context — a way in, not a report.
"""
from __future__ import annotations
from shonar.services.ai import SummaryResult
SYSTEM_PROMPT = (
"You summarize voice recordings for the speaker's own later reference. "
"Reply with JSON only, exactly these keys: "
'{"short": "1-2 sentences", "detailed": "a faithful paragraph", '
'"key_points": [], "decisions": [], "action_items": [], "questions": []}. '
"Empty arrays when absent. Never invent names, dates, or commitments "
"not stated in the transcript."
)
MAX_TRANSCRIPT_CHARS = 12_000
def build_user_message(transcript: str, title: str | None,
tone: str | None = None) -> str:
text = transcript[:MAX_TRANSCRIPT_CHARS]
if len(transcript) > MAX_TRANSCRIPT_CHARS:
text += f"\n\n[truncated from {len(transcript)} chars]"
head = f'Title: "{title}"\n\n' if title else ""
msg = head + "Transcript:\n" + text
if tone:
# Same JSON contract and same fidelity rules — only the voice changes.
msg += (
f"\n\nWrite every field of the JSON in a {tone} tone of voice. "
"Stay faithful to the transcript: the tone colors the wording, "
"never the facts."
)
return msg
# Progress from REAL completed work units only — never a timer, never an
# elapsed-time estimate, never a char-count guess against a made-up output
# size. The summary contract is a fixed JSON shape (six keys), and JSON
# structure is parseable from the stream prefix, so every point the bar
# advances corresponds to a unit of output the model has actually
# finished: the first token, each closed contract key, each closed array
# item, the closing brace. The un-finishable part — the thinking phase
# and the interior of an open value — honestly earns zero points: no
# provider (llama.cpp, vLLM, Ollama) exposes a token target while
# streaming, so any "continuous" percent over that span would be fake.
# Milestones: 0 queued · 5 first token · 5→90 six contract keys
# (arrays advance per closed item) · 95 root closed/parsed · 100 stored.
_CONTRACT_FIRST_TOKEN = 5
_CONTRACT_ROOT = 95
# Fixed per-key budget (5..90 band). detailed is contractually the
# longest field; arrays share a band with per-item steps inside.
_KEY_POINTS_BUDGET = {
"short": 12,
"detailed": 20,
"key_points": 13,
"decisions": 13,
"action_items": 13,
"questions": 13,
}
_ARR_OPEN, _ARR_ITEM, _ARR_ITEM_MAX, _ARR_CLOSE = 3, 2, 7, 4
class ContractProgress:
"""Streaming-JSON milestone tracker. feed(text_prefix) accepts the
accumulated reply content (repeated prefixes are fine — scanning
resumes where it left off) and returns a new pct, or None when
nothing new completed. Monotonic by construction."""
def __init__(self) -> None:
self._pos = 0 # scan offset into the prefix
self._depth = 0 # {} and [] nesting
self._in_str = False
self._esc = False
self._str_buf: list[str] = []
self._cur_key: str | None = None # key whose value we're in
self._at_key_slot = True # next depth-1 string is a key
self._items_open = 0 # items closed in current array
self._key_score = 0 # points earned for current key
self._earned = 0 # points above the first-token step
self._done_keys: set[str] = set()
self._reported = 0
self._root_closed = False
def _emit(self, pct: int) -> int | None:
pct = min(99, max(0, pct))
if pct > self._reported:
self._reported = pct
return pct
return None
def _award(self, pts: int, cap: int) -> None:
"""Add points to the current key, capped at that key's budget."""
grant = min(pts, cap - self._key_score)
if grant > 0:
self._key_score += grant
self._earned += grant
def feed(self, prefix: str) -> list[int]:
"""Scan the new tail of the accumulated prefix; return every
newly earned pct in order (possibly empty)."""
emits: list[int] = []
if self._reported == 0:
got = self._emit(_CONTRACT_FIRST_TOKEN)
if got is not None:
emits.append(got)
text = prefix
while self._pos < len(text):
c = text[self._pos]
self._pos += 1
if self._in_str:
if self._esc:
self._esc = False
elif c == "\\":
self._esc = True
elif c == '"':
self._in_str = False
s = "".join(self._str_buf)
self._str_buf = []
if self._depth == 1:
# Depth-1 strings alternate key, value, key, ...
# Slot tracking (not lookahead) so a key closing
# one chunk before its ':' is not misread.
if self._at_key_slot:
self._cur_key = s
self._key_score = 0
self._at_key_slot = False
elif self._cur_key in ("short", "detailed"):
self._award(_KEY_POINTS_BUDGET[self._cur_key],
_KEY_POINTS_BUDGET[self._cur_key])
self._done_keys.add(self._cur_key)
self._cur_key = None
got = self._emit(_CONTRACT_FIRST_TOKEN + self._earned)
if got is not None:
emits.append(got)
self._at_key_slot = True
elif self._depth == 2:
# a closed array item (arrays hold strings)
if self._cur_key in _KEY_POINTS_BUDGET:
self._items_open += 1
remaining = _ARR_ITEM_MAX \
- _ARR_ITEM * (self._items_open - 1)
self._award(min(_ARR_ITEM, max(0, remaining)),
_KEY_POINTS_BUDGET[self._cur_key])
got = self._emit(_CONTRACT_FIRST_TOKEN + self._earned)
if got is not None:
emits.append(got)
else:
self._str_buf.append(c)
continue
if c == '"':
self._in_str = True
self._str_buf = []
elif c in "{[":
self._depth += 1
if c == "[" and self._cur_key:
self._items_open = 0
self._award(_ARR_OPEN, _KEY_POINTS_BUDGET[self._cur_key])
got = self._emit(_CONTRACT_FIRST_TOKEN + self._earned)
if got is not None:
emits.append(got)
elif c in "}]":
self._depth -= 1
if self._depth == 1:
if c == "]":
if self._cur_key in _KEY_POINTS_BUDGET:
self._award(_ARR_CLOSE,
_KEY_POINTS_BUDGET[self._cur_key])
self._done_keys.add(self._cur_key)
self._cur_key = None
got = self._emit(_CONTRACT_FIRST_TOKEN + self._earned)
if got is not None:
emits.append(got)
self._at_key_slot = True
elif self._depth == 0:
self._root_closed = True
got = self._emit(_CONTRACT_ROOT)
if got is not None:
emits.append(got)
elif c == "," and self._depth == 1:
self._at_key_slot = True
return emits
def make_progress_ticker(on_progress):
"""Wrap an optional on_progress callback into a feed(prefix) sink.
Receives the accumulated reply *content* (not reasoning) as it
streams and reports 5..95 strictly from completed contract units —
see ContractProgress for what each point means. Silent no-op without
a callback; failures never disturb the summary itself."""
if on_progress is None:
return lambda prefix: None
tracker = ContractProgress()
def feed(prefix: str) -> None:
try: # noqa: SIM105 — swallow deliberately: progress is display state
pcts = tracker.feed(prefix or "")
except Exception: # display state only
return
for pct in pcts:
on_progress(pct)
return feed
def parse_summary(data: object, model: str) -> SummaryResult:
if not isinstance(data, dict):
return SummaryResult(model=model)
return SummaryResult.from_dict(data, model=model)