Summarize: LAN primary with local-Ollama rescue + in-app failure guidance
- Engine: when the primary summarizer is out of retries (or misconfigured), run_summarize now finishes the job on the configured rescue provider (SHONAR_LLM_FALLBACK_*), tags the summary with the provider that wrote it, and stores a human note in job.error; success clears stale notes. - App (auto/lan): passes Ollama as the rescue provider when it is up. - Detail screen: shows the rescue-swap note in plain words, a red 'Summary failed' line with Settings -> Summarizer fix instructions and a Retry summary button on hard failure. - Settings copy explains the fallback. 3 new pytest cases (14/14 pass); live E2E on 2026-09-18: sarcastic summary v6 via LAN on attempt 2.
This commit is contained in:
parent
cb0492fb3e
commit
3061629dc5
6 changed files with 172 additions and 8 deletions
|
|
@ -358,6 +358,15 @@ class DesktopState(private val appDir: File = defaultAppDir()) {
|
||||||
env["SHONAR_LLM_BASE_URL"] = LAN_LLM_BASE_URL
|
env["SHONAR_LLM_BASE_URL"] = LAN_LLM_BASE_URL
|
||||||
env["SHONAR_LLM_MODEL"] = LAN_LLM_MODEL
|
env["SHONAR_LLM_MODEL"] = LAN_LLM_MODEL
|
||||||
env["SHONAR_LLM_API_KEY"] = lanKey
|
env["SHONAR_LLM_API_KEY"] = lanKey
|
||||||
|
// Rescue summarizer: the LAN GPU intermittently replies
|
||||||
|
// with garbage; when a summarize job exhausts its retries
|
||||||
|
// the engine finishes it on local Ollama instead of
|
||||||
|
// failing, and the detail screen names the swap.
|
||||||
|
if (ollamaUp()) {
|
||||||
|
env["SHONAR_LLM_FALLBACK_PROVIDER"] = "ollama"
|
||||||
|
env["SHONAR_LLM_FALLBACK_BASE_URL"] = "http://127.0.0.1:11434"
|
||||||
|
env["SHONAR_LLM_FALLBACK_MODEL"] = ollamaModel()
|
||||||
|
}
|
||||||
} else if (ollamaUp()) {
|
} else if (ollamaUp()) {
|
||||||
env["SHONAR_LLM_PROVIDER"] = "ollama"
|
env["SHONAR_LLM_PROVIDER"] = "ollama"
|
||||||
env["SHONAR_LLM_BASE_URL"] = "http://127.0.0.1:11434"
|
env["SHONAR_LLM_BASE_URL"] = "http://127.0.0.1:11434"
|
||||||
|
|
|
||||||
|
|
@ -796,6 +796,35 @@ fun DetailScreen(state: DesktopState) {
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// Rescue-swap note: the engine succeeded but had to switch
|
||||||
|
// summarizers mid-job. Tell the user in plain words, not the raw
|
||||||
|
// provider ids.
|
||||||
|
sJob?.takeIf { it.status == "succeeded" && it.error != null }?.let {
|
||||||
|
val note = it.error!!
|
||||||
|
.replace("openai_compat", "LAN server (H200)")
|
||||||
|
.replace("ollama", "this laptop (Ollama)")
|
||||||
|
Text(note, style = MaterialTheme.typography.bodySmall,
|
||||||
|
color = MaterialTheme.colorScheme.tertiary)
|
||||||
|
Text("You can change the summarizer in Settings → Summarizer.",
|
||||||
|
style = MaterialTheme.typography.bodySmall,
|
||||||
|
color = MaterialTheme.colorScheme.onSurfaceVariant)
|
||||||
|
}
|
||||||
|
// Summarize failed even after any rescue: say what broke and how to
|
||||||
|
// fix it from inside the app.
|
||||||
|
sJob?.takeIf { it.status == "failed" }?.let {
|
||||||
|
Text("Summary failed: " + (it.error ?: "unknown error")
|
||||||
|
.replace("openai_compat", "LAN server (H200)")
|
||||||
|
.replace("ollama", "this laptop (Ollama)"),
|
||||||
|
style = MaterialTheme.typography.bodySmall,
|
||||||
|
color = MaterialTheme.colorScheme.error)
|
||||||
|
Text("Fix: open Settings → Summarizer and pick \"This laptop " +
|
||||||
|
"(Ollama)\", then press Retry summary below.",
|
||||||
|
style = MaterialTheme.typography.bodySmall,
|
||||||
|
color = MaterialTheme.colorScheme.onSurfaceVariant)
|
||||||
|
if (detail.busy == null) {
|
||||||
|
Button({ state.summarize() }) { Text("Retry summary") }
|
||||||
|
}
|
||||||
|
}
|
||||||
detail.error?.let {
|
detail.error?.let {
|
||||||
Text(it, color = MaterialTheme.colorScheme.error)
|
Text(it, color = MaterialTheme.colorScheme.error)
|
||||||
if (tJob?.status == "failed") {
|
if (tJob?.status == "failed") {
|
||||||
|
|
@ -1167,11 +1196,14 @@ fun SettingsScreen(state: DesktopState) {
|
||||||
}
|
}
|
||||||
Text(when (summarizer) {
|
Text(when (summarizer) {
|
||||||
"lan" -> "Best quality; shares the H200 with other clients — " +
|
"lan" -> "Best quality; shares the H200 with other clients — " +
|
||||||
"summaries may wait in its queue before starting."
|
"summaries may wait in its queue before starting. If it " +
|
||||||
|
"keeps failing, the job finishes on this laptop's Ollama " +
|
||||||
|
"and this screen says so."
|
||||||
"local" -> "Private and never queued behind other users; " +
|
"local" -> "Private and never queued behind other users; " +
|
||||||
"smaller model, so summaries are plainer."
|
"smaller model, so summaries are plainer."
|
||||||
else -> "Uses the LAN server when its key is available, " +
|
else -> "Uses the LAN server when its key is available " +
|
||||||
"otherwise this laptop's Ollama."
|
"(falling back to this laptop's Ollama on repeated " +
|
||||||
|
"failures), otherwise Ollama straight away."
|
||||||
},
|
},
|
||||||
style = MaterialTheme.typography.bodySmall,
|
style = MaterialTheme.typography.bodySmall,
|
||||||
color = MaterialTheme.colorScheme.onSurfaceVariant)
|
color = MaterialTheme.colorScheme.onSurfaceVariant)
|
||||||
|
|
|
||||||
|
|
@ -90,6 +90,14 @@ class Settings(BaseSettings):
|
||||||
llm_base_url: str = ""
|
llm_base_url: str = ""
|
||||||
llm_api_key: str = Field(default="", repr=False)
|
llm_api_key: str = Field(default="", repr=False)
|
||||||
|
|
||||||
|
# Rescue summarizer tried when the PRIMARY one fails mid-job (desktop:
|
||||||
|
# LAN GPU server primary, local Ollama fallback). "none" disables the
|
||||||
|
# fallback; the primary's error then stands as today.
|
||||||
|
llm_fallback_provider: str = "none" # none | openai_compat | ollama
|
||||||
|
llm_fallback_model: str = ""
|
||||||
|
llm_fallback_base_url: str = ""
|
||||||
|
llm_fallback_api_key: str = Field(default="", repr=False)
|
||||||
|
|
||||||
# Search needs no setting: services/search.py picks its path by SQL
|
# Search needs no setting: services/search.py picks its path by SQL
|
||||||
# dialect (SQLite substring scan / Postgres tsvector). A Meilisearch or
|
# dialect (SQLite substring scan / Postgres tsvector). A Meilisearch or
|
||||||
# OpenSearch backend would replace that module, not add a config knob.
|
# OpenSearch backend would replace that module, not add a config knob.
|
||||||
|
|
|
||||||
|
|
@ -160,3 +160,32 @@ def get_llm_provider(settings: Settings) -> LlmProvider | None:
|
||||||
model=settings.llm_model,
|
model=settings.llm_model,
|
||||||
)
|
)
|
||||||
raise ProviderConfigError(f"Unknown LLM provider: {settings.llm_provider!r}")
|
raise ProviderConfigError(f"Unknown LLM provider: {settings.llm_provider!r}")
|
||||||
|
|
||||||
|
|
||||||
|
def get_llm_fallback_provider(settings: Settings) -> LlmProvider | None:
|
||||||
|
"""Rescue provider tried when the primary summarize fails mid-job.
|
||||||
|
|
||||||
|
None means "no fallback configured" — the primary's error stands.
|
||||||
|
Same kinds as get_llm_provider, built from the llm_fallback_* settings.
|
||||||
|
"""
|
||||||
|
kind = settings.llm_fallback_provider.strip().lower()
|
||||||
|
if kind in ("", "none"):
|
||||||
|
return None
|
||||||
|
if kind == "openai_compat":
|
||||||
|
from shonar.services.ai.openai_compat import OpenAICompatProvider
|
||||||
|
|
||||||
|
return OpenAICompatProvider(
|
||||||
|
base_url=settings.llm_fallback_base_url,
|
||||||
|
model=settings.llm_fallback_model,
|
||||||
|
api_key=settings.llm_fallback_api_key,
|
||||||
|
)
|
||||||
|
if kind == "ollama":
|
||||||
|
from shonar.services.ai.ollama import OllamaProvider
|
||||||
|
|
||||||
|
return OllamaProvider(
|
||||||
|
base_url=settings.llm_fallback_base_url,
|
||||||
|
model=settings.llm_fallback_model,
|
||||||
|
)
|
||||||
|
raise ProviderConfigError(
|
||||||
|
f"Unknown LLM fallback provider: {settings.llm_fallback_provider!r}"
|
||||||
|
)
|
||||||
|
|
|
||||||
|
|
@ -39,7 +39,9 @@ from shonar.db.models import (
|
||||||
)
|
)
|
||||||
from shonar.services.ai import (
|
from shonar.services.ai import (
|
||||||
AIError,
|
AIError,
|
||||||
|
ProviderConfigError,
|
||||||
ProviderTransientError,
|
ProviderTransientError,
|
||||||
|
get_llm_fallback_provider,
|
||||||
get_llm_provider,
|
get_llm_provider,
|
||||||
get_transcription_provider,
|
get_transcription_provider,
|
||||||
)
|
)
|
||||||
|
|
@ -479,19 +481,48 @@ async def run_summarize(ctx: dict, recording_id: str) -> None:
|
||||||
job.stage = "summarizing"
|
job.stage = "summarizing"
|
||||||
rec.processing_status = ProcessingStatus.processing
|
rec.processing_status = ProcessingStatus.processing
|
||||||
await session.commit() # visible before the long LLM call
|
await session.commit() # visible before the long LLM call
|
||||||
|
fallback_note: str | None = None
|
||||||
|
result_provider = provider.name # overwritten only on the rescue path
|
||||||
try:
|
try:
|
||||||
result = await provider.summarize(
|
result = await provider.summarize(
|
||||||
text, title=rec.title, tone=job.tone,
|
text, title=rec.title, tone=job.tone,
|
||||||
on_progress=_async_progress_writer(job.id))
|
on_progress=_async_progress_writer(job.id))
|
||||||
except AIError as e:
|
except AIError as e:
|
||||||
await _fail(session, rec, job, str(e), ctx, e)
|
# Transient failures keep their normal retry budget first (the
|
||||||
await session.commit()
|
# LAN GPU often recovers on try 2); the rescue summarizer runs
|
||||||
return
|
# when the primary is OUT of tries or misconfigured, and only
|
||||||
await store_summary(session, rec, result.to_dict(), provider.name,
|
# if one is configured. The summary then comes from the
|
||||||
|
# fallback and job.error carries a note for the UI.
|
||||||
|
retryable = (isinstance(e, ProviderTransientError)
|
||||||
|
and _job_try(ctx) < MAX_TRIES)
|
||||||
|
# (result, provider_name, note) when the rescue succeeds.
|
||||||
|
rescue: tuple[object, str, str] | None = None
|
||||||
|
if not retryable:
|
||||||
|
fallback = get_llm_fallback_provider(settings)
|
||||||
|
if fallback is not None:
|
||||||
|
try:
|
||||||
|
fb_result = await fallback.summarize(
|
||||||
|
text, title=rec.title, tone=job.tone,
|
||||||
|
on_progress=_async_progress_writer(job.id))
|
||||||
|
except AIError as fe:
|
||||||
|
logger.warning(
|
||||||
|
"summarize fallback (%s) also failed: %s",
|
||||||
|
fallback.name, fe)
|
||||||
|
else:
|
||||||
|
rescue = (fb_result, fallback.name, (
|
||||||
|
f'primary "{provider.name}" failed ({e}). '
|
||||||
|
f'This summary was written by "{fallback.name}".'))
|
||||||
|
if rescue is None:
|
||||||
|
await _fail(session, rec, job, str(e), ctx, e)
|
||||||
|
await session.commit()
|
||||||
|
return
|
||||||
|
result, result_provider, fallback_note = rescue
|
||||||
|
await store_summary(session, rec, result.to_dict(), result_provider,
|
||||||
result.model, tone=job.tone)
|
result.model, tone=job.tone)
|
||||||
job.status = JobStatus.succeeded
|
job.status = JobStatus.succeeded
|
||||||
job.stage = None
|
job.stage = None
|
||||||
job.progress = 100
|
job.progress = 100
|
||||||
|
job.error = fallback_note # None on the clean path: clears stale notes
|
||||||
job.finished_at = utcnow()
|
job.finished_at = utcnow()
|
||||||
rec.processing_status = ProcessingStatus.completed
|
rec.processing_status = ProcessingStatus.completed
|
||||||
rec.processing_error = None
|
rec.processing_error = None
|
||||||
|
|
|
||||||
|
|
@ -110,11 +110,13 @@ class FakeLlm:
|
||||||
_UNSET = object()
|
_UNSET = object()
|
||||||
|
|
||||||
|
|
||||||
def use_fakes(monkeypatch, stt=_UNSET, llm=_UNSET):
|
def use_fakes(monkeypatch, stt=_UNSET, llm=_UNSET, fallback=None):
|
||||||
tprov = FakeTranscriber() if stt is _UNSET else stt
|
tprov = FakeTranscriber() if stt is _UNSET else stt
|
||||||
lprov = FakeLlm() if llm is _UNSET else llm
|
lprov = FakeLlm() if llm is _UNSET else llm
|
||||||
monkeypatch.setattr(processing, "get_transcription_provider", lambda settings: tprov)
|
monkeypatch.setattr(processing, "get_transcription_provider", lambda settings: tprov)
|
||||||
monkeypatch.setattr(processing, "get_llm_provider", lambda settings: lprov)
|
monkeypatch.setattr(processing, "get_llm_provider", lambda settings: lprov)
|
||||||
|
monkeypatch.setattr(processing, "get_llm_fallback_provider",
|
||||||
|
lambda settings: fallback)
|
||||||
|
|
||||||
|
|
||||||
async def jobs_for(recording_id):
|
async def jobs_for(recording_id):
|
||||||
|
|
@ -409,3 +411,56 @@ async def test_ai_endpoints_enforce_ownership(client, monkeypatch):
|
||||||
for path in ("transcript", "summary", "jobs"):
|
for path in ("transcript", "summary", "jobs"):
|
||||||
r = await client.get(f"/api/v1/recordings/{rec['id']}/{path}", headers=h)
|
r = await client.get(f"/api/v1/recordings/{rec['id']}/{path}", headers=h)
|
||||||
assert r.status_code == 404, path
|
assert r.status_code == 404, path
|
||||||
|
|
||||||
|
|
||||||
|
# --- summarize fallback to the rescue summarizer --------------------------------
|
||||||
|
|
||||||
|
async def test_summarize_fallback_rescues_after_primary_exhausted(client, monkeypatch):
|
||||||
|
primary = FakeLlm(fail=ProviderTransientError("LLM reply was not JSON."))
|
||||||
|
fb = FakeLlm()
|
||||||
|
fb.name = "fake-ollama"
|
||||||
|
use_fakes(monkeypatch, FakeTranscriber(), primary, fallback=fb)
|
||||||
|
token = await user_tokens(client, email="m7-fb1@example.com")
|
||||||
|
rec = await upload_recording(client, token, client_id="m7-fb-1")
|
||||||
|
await processing.run_transcribe({}, rec["id"])
|
||||||
|
# Last try: primary out of budget -> rescue runs and the note lands in job.error.
|
||||||
|
await processing.run_summarize({"job_try": 3}, rec["id"])
|
||||||
|
job = (await jobs_for(rec["id"]))[JobType.summarize]
|
||||||
|
assert job.status == JobStatus.succeeded
|
||||||
|
assert "fake-llm" in (job.error or "") and "fake-ollama" in (job.error or "")
|
||||||
|
h = {"Authorization": f"Bearer {token}"}
|
||||||
|
s = await client.get(f"/api/v1/recordings/{rec['id']}/summary", headers=h)
|
||||||
|
assert s.status_code == 200
|
||||||
|
assert s.json()["provider"] == "fake-ollama"
|
||||||
|
assert fb.seen == ["hello world from the meeting"]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_summarize_fallback_also_fails_marks_job_failed(client, monkeypatch):
|
||||||
|
err = ProviderTransientError("LLM reply was not JSON.")
|
||||||
|
fb = FakeLlm(fail=ProviderTransientError("ollama down"))
|
||||||
|
fb.name = "fake-ollama"
|
||||||
|
use_fakes(monkeypatch, FakeTranscriber(), FakeLlm(fail=err), fallback=fb)
|
||||||
|
token = await user_tokens(client, email="m7-fb2@example.com")
|
||||||
|
rec = await upload_recording(client, token, client_id="m7-fb-2")
|
||||||
|
await processing.run_transcribe({}, rec["id"])
|
||||||
|
await processing.run_summarize({"job_try": 3}, rec["id"])
|
||||||
|
job = (await jobs_for(rec["id"]))[JobType.summarize]
|
||||||
|
assert job.status == JobStatus.failed
|
||||||
|
assert "not JSON" in job.error # the primary's error is what we surface
|
||||||
|
assert fb.seen == ["hello world from the meeting"] # fallback was tried
|
||||||
|
|
||||||
|
|
||||||
|
async def test_summarize_keeps_retry_budget_before_fallback(client, monkeypatch):
|
||||||
|
# Try 1 of 3 transient: normal arq retry wins, fallback stays untouched.
|
||||||
|
fb = FakeLlm()
|
||||||
|
fb.name = "fake-ollama"
|
||||||
|
use_fakes(monkeypatch, FakeTranscriber(),
|
||||||
|
FakeLlm(fail=ProviderTransientError("boom")), fallback=fb)
|
||||||
|
token = await user_tokens(client, email="m7-fb3@example.com")
|
||||||
|
rec = await upload_recording(client, token, client_id="m7-fb-3")
|
||||||
|
await processing.run_transcribe({}, rec["id"])
|
||||||
|
with pytest.raises(ProviderTransientError):
|
||||||
|
await processing.run_summarize({"job_try": 1}, rec["id"])
|
||||||
|
job = (await jobs_for(rec["id"]))[JobType.summarize]
|
||||||
|
assert job.status == JobStatus.queued
|
||||||
|
assert fb.seen == []
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue