diff --git a/app/src/main/kotlin/com/shonar/desktop/DesktopState.kt b/app/src/main/kotlin/com/shonar/desktop/DesktopState.kt index 6898eba..53dccbf 100644 --- a/app/src/main/kotlin/com/shonar/desktop/DesktopState.kt +++ b/app/src/main/kotlin/com/shonar/desktop/DesktopState.kt @@ -358,6 +358,15 @@ class DesktopState(private val appDir: File = defaultAppDir()) { env["SHONAR_LLM_BASE_URL"] = LAN_LLM_BASE_URL env["SHONAR_LLM_MODEL"] = LAN_LLM_MODEL env["SHONAR_LLM_API_KEY"] = lanKey + // Rescue summarizer: the LAN GPU intermittently replies + // with garbage; when a summarize job exhausts its retries + // the engine finishes it on local Ollama instead of + // failing, and the detail screen names the swap. + if (ollamaUp()) { + env["SHONAR_LLM_FALLBACK_PROVIDER"] = "ollama" + env["SHONAR_LLM_FALLBACK_BASE_URL"] = "http://127.0.0.1:11434" + env["SHONAR_LLM_FALLBACK_MODEL"] = ollamaModel() + } } else if (ollamaUp()) { env["SHONAR_LLM_PROVIDER"] = "ollama" env["SHONAR_LLM_BASE_URL"] = "http://127.0.0.1:11434" diff --git a/app/src/main/kotlin/com/shonar/desktop/Screens.kt b/app/src/main/kotlin/com/shonar/desktop/Screens.kt index d0fe6bf..7648aef 100644 --- a/app/src/main/kotlin/com/shonar/desktop/Screens.kt +++ b/app/src/main/kotlin/com/shonar/desktop/Screens.kt @@ -796,6 +796,35 @@ fun DetailScreen(state: DesktopState) { ) } } + // Rescue-swap note: the engine succeeded but had to switch + // summarizers mid-job. Tell the user in plain words, not the raw + // provider ids. + sJob?.takeIf { it.status == "succeeded" && it.error != null }?.let { + val note = it.error!! + .replace("openai_compat", "LAN server (H200)") + .replace("ollama", "this laptop (Ollama)") + Text(note, style = MaterialTheme.typography.bodySmall, + color = MaterialTheme.colorScheme.tertiary) + Text("You can change the summarizer in Settings → Summarizer.", + style = MaterialTheme.typography.bodySmall, + color = MaterialTheme.colorScheme.onSurfaceVariant) + } + // Summarize failed even after any rescue: say what broke and how to + // fix it from inside the app. + sJob?.takeIf { it.status == "failed" }?.let { + Text("Summary failed: " + (it.error ?: "unknown error") + .replace("openai_compat", "LAN server (H200)") + .replace("ollama", "this laptop (Ollama)"), + style = MaterialTheme.typography.bodySmall, + color = MaterialTheme.colorScheme.error) + Text("Fix: open Settings → Summarizer and pick \"This laptop " + + "(Ollama)\", then press Retry summary below.", + style = MaterialTheme.typography.bodySmall, + color = MaterialTheme.colorScheme.onSurfaceVariant) + if (detail.busy == null) { + Button({ state.summarize() }) { Text("Retry summary") } + } + } detail.error?.let { Text(it, color = MaterialTheme.colorScheme.error) if (tJob?.status == "failed") { @@ -1167,11 +1196,14 @@ fun SettingsScreen(state: DesktopState) { } Text(when (summarizer) { "lan" -> "Best quality; shares the H200 with other clients — " + - "summaries may wait in its queue before starting." + "summaries may wait in its queue before starting. If it " + + "keeps failing, the job finishes on this laptop's Ollama " + + "and this screen says so." "local" -> "Private and never queued behind other users; " + "smaller model, so summaries are plainer." - else -> "Uses the LAN server when its key is available, " + - "otherwise this laptop's Ollama." + else -> "Uses the LAN server when its key is available " + + "(falling back to this laptop's Ollama on repeated " + + "failures), otherwise Ollama straight away." }, style = MaterialTheme.typography.bodySmall, color = MaterialTheme.colorScheme.onSurfaceVariant) diff --git a/backend/shonar/core/config.py b/backend/shonar/core/config.py index 9f69a6c..8f2d22e 100644 --- a/backend/shonar/core/config.py +++ b/backend/shonar/core/config.py @@ -90,6 +90,14 @@ class Settings(BaseSettings): llm_base_url: str = "" llm_api_key: str = Field(default="", repr=False) + # Rescue summarizer tried when the PRIMARY one fails mid-job (desktop: + # LAN GPU server primary, local Ollama fallback). "none" disables the + # fallback; the primary's error then stands as today. + llm_fallback_provider: str = "none" # none | openai_compat | ollama + llm_fallback_model: str = "" + llm_fallback_base_url: str = "" + llm_fallback_api_key: str = Field(default="", repr=False) + # Search needs no setting: services/search.py picks its path by SQL # dialect (SQLite substring scan / Postgres tsvector). A Meilisearch or # OpenSearch backend would replace that module, not add a config knob. diff --git a/backend/shonar/services/ai/__init__.py b/backend/shonar/services/ai/__init__.py index 0cd13d9..04d6145 100644 --- a/backend/shonar/services/ai/__init__.py +++ b/backend/shonar/services/ai/__init__.py @@ -160,3 +160,32 @@ def get_llm_provider(settings: Settings) -> LlmProvider | None: model=settings.llm_model, ) raise ProviderConfigError(f"Unknown LLM provider: {settings.llm_provider!r}") + + +def get_llm_fallback_provider(settings: Settings) -> LlmProvider | None: + """Rescue provider tried when the primary summarize fails mid-job. + + None means "no fallback configured" — the primary's error stands. + Same kinds as get_llm_provider, built from the llm_fallback_* settings. + """ + kind = settings.llm_fallback_provider.strip().lower() + if kind in ("", "none"): + return None + if kind == "openai_compat": + from shonar.services.ai.openai_compat import OpenAICompatProvider + + return OpenAICompatProvider( + base_url=settings.llm_fallback_base_url, + model=settings.llm_fallback_model, + api_key=settings.llm_fallback_api_key, + ) + if kind == "ollama": + from shonar.services.ai.ollama import OllamaProvider + + return OllamaProvider( + base_url=settings.llm_fallback_base_url, + model=settings.llm_fallback_model, + ) + raise ProviderConfigError( + f"Unknown LLM fallback provider: {settings.llm_fallback_provider!r}" + ) diff --git a/backend/shonar/services/processing.py b/backend/shonar/services/processing.py index 912ff1e..bebfef7 100644 --- a/backend/shonar/services/processing.py +++ b/backend/shonar/services/processing.py @@ -39,7 +39,9 @@ from shonar.db.models import ( ) from shonar.services.ai import ( AIError, + ProviderConfigError, ProviderTransientError, + get_llm_fallback_provider, get_llm_provider, get_transcription_provider, ) @@ -479,19 +481,48 @@ async def run_summarize(ctx: dict, recording_id: str) -> None: job.stage = "summarizing" rec.processing_status = ProcessingStatus.processing await session.commit() # visible before the long LLM call + fallback_note: str | None = None + result_provider = provider.name # overwritten only on the rescue path try: result = await provider.summarize( text, title=rec.title, tone=job.tone, on_progress=_async_progress_writer(job.id)) except AIError as e: - await _fail(session, rec, job, str(e), ctx, e) - await session.commit() - return - await store_summary(session, rec, result.to_dict(), provider.name, + # Transient failures keep their normal retry budget first (the + # LAN GPU often recovers on try 2); the rescue summarizer runs + # when the primary is OUT of tries or misconfigured, and only + # if one is configured. The summary then comes from the + # fallback and job.error carries a note for the UI. + retryable = (isinstance(e, ProviderTransientError) + and _job_try(ctx) < MAX_TRIES) + # (result, provider_name, note) when the rescue succeeds. + rescue: tuple[object, str, str] | None = None + if not retryable: + fallback = get_llm_fallback_provider(settings) + if fallback is not None: + try: + fb_result = await fallback.summarize( + text, title=rec.title, tone=job.tone, + on_progress=_async_progress_writer(job.id)) + except AIError as fe: + logger.warning( + "summarize fallback (%s) also failed: %s", + fallback.name, fe) + else: + rescue = (fb_result, fallback.name, ( + f'primary "{provider.name}" failed ({e}). ' + f'This summary was written by "{fallback.name}".')) + if rescue is None: + await _fail(session, rec, job, str(e), ctx, e) + await session.commit() + return + result, result_provider, fallback_note = rescue + await store_summary(session, rec, result.to_dict(), result_provider, result.model, tone=job.tone) job.status = JobStatus.succeeded job.stage = None job.progress = 100 + job.error = fallback_note # None on the clean path: clears stale notes job.finished_at = utcnow() rec.processing_status = ProcessingStatus.completed rec.processing_error = None diff --git a/backend/tests/test_ai_pipeline.py b/backend/tests/test_ai_pipeline.py index 71dd24d..8b0d56d 100644 --- a/backend/tests/test_ai_pipeline.py +++ b/backend/tests/test_ai_pipeline.py @@ -110,11 +110,13 @@ class FakeLlm: _UNSET = object() -def use_fakes(monkeypatch, stt=_UNSET, llm=_UNSET): +def use_fakes(monkeypatch, stt=_UNSET, llm=_UNSET, fallback=None): tprov = FakeTranscriber() if stt is _UNSET else stt lprov = FakeLlm() if llm is _UNSET else llm monkeypatch.setattr(processing, "get_transcription_provider", lambda settings: tprov) monkeypatch.setattr(processing, "get_llm_provider", lambda settings: lprov) + monkeypatch.setattr(processing, "get_llm_fallback_provider", + lambda settings: fallback) async def jobs_for(recording_id): @@ -409,3 +411,56 @@ async def test_ai_endpoints_enforce_ownership(client, monkeypatch): for path in ("transcript", "summary", "jobs"): r = await client.get(f"/api/v1/recordings/{rec['id']}/{path}", headers=h) assert r.status_code == 404, path + + +# --- summarize fallback to the rescue summarizer -------------------------------- + +async def test_summarize_fallback_rescues_after_primary_exhausted(client, monkeypatch): + primary = FakeLlm(fail=ProviderTransientError("LLM reply was not JSON.")) + fb = FakeLlm() + fb.name = "fake-ollama" + use_fakes(monkeypatch, FakeTranscriber(), primary, fallback=fb) + token = await user_tokens(client, email="m7-fb1@example.com") + rec = await upload_recording(client, token, client_id="m7-fb-1") + await processing.run_transcribe({}, rec["id"]) + # Last try: primary out of budget -> rescue runs and the note lands in job.error. + await processing.run_summarize({"job_try": 3}, rec["id"]) + job = (await jobs_for(rec["id"]))[JobType.summarize] + assert job.status == JobStatus.succeeded + assert "fake-llm" in (job.error or "") and "fake-ollama" in (job.error or "") + h = {"Authorization": f"Bearer {token}"} + s = await client.get(f"/api/v1/recordings/{rec['id']}/summary", headers=h) + assert s.status_code == 200 + assert s.json()["provider"] == "fake-ollama" + assert fb.seen == ["hello world from the meeting"] + + +async def test_summarize_fallback_also_fails_marks_job_failed(client, monkeypatch): + err = ProviderTransientError("LLM reply was not JSON.") + fb = FakeLlm(fail=ProviderTransientError("ollama down")) + fb.name = "fake-ollama" + use_fakes(monkeypatch, FakeTranscriber(), FakeLlm(fail=err), fallback=fb) + token = await user_tokens(client, email="m7-fb2@example.com") + rec = await upload_recording(client, token, client_id="m7-fb-2") + await processing.run_transcribe({}, rec["id"]) + await processing.run_summarize({"job_try": 3}, rec["id"]) + job = (await jobs_for(rec["id"]))[JobType.summarize] + assert job.status == JobStatus.failed + assert "not JSON" in job.error # the primary's error is what we surface + assert fb.seen == ["hello world from the meeting"] # fallback was tried + + +async def test_summarize_keeps_retry_budget_before_fallback(client, monkeypatch): + # Try 1 of 3 transient: normal arq retry wins, fallback stays untouched. + fb = FakeLlm() + fb.name = "fake-ollama" + use_fakes(monkeypatch, FakeTranscriber(), + FakeLlm(fail=ProviderTransientError("boom")), fallback=fb) + token = await user_tokens(client, email="m7-fb3@example.com") + rec = await upload_recording(client, token, client_id="m7-fb-3") + await processing.run_transcribe({}, rec["id"]) + with pytest.raises(ProviderTransientError): + await processing.run_summarize({"job_try": 1}, rec["id"]) + job = (await jobs_for(rec["id"]))[JobType.summarize] + assert job.status == JobStatus.queued + assert fb.seen == []