Standalone Shonar Desktop: vendor portable sources + local engine; decouple from ~/Projects/Shonar
- shared/ = portable Android-origin sources vendored from deferred/desktop-server (app/build.gradle.kts srcDir repointed; PlaybackController.kt excluded as Android-only) - backend/ = bundled-lite engine (SQLite + inline queue); .venv symlinked from the old checkout, PYTHONPATH pins THIS backend's code over any editable install - repoRoot() resolves this project dir (env SHONAR_REPO still wins); desktop-dev.sh watches shared/ + backend/ - Verified: :app:compileKotlin + :app:test green (23 tests); engine boots on :8010, self-migrates, /healthz ok
This commit is contained in:
commit
76c867fca4
136 changed files with 21099 additions and 0 deletions
78
backend/shonar/services/ai/ollama.py
Normal file
78
backend/shonar/services/ai/ollama.py
Normal file
|
|
@ -0,0 +1,78 @@
|
|||
"""Summaries via a local Ollama server (`/api/chat`, JSON mode)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import AsyncIterator
|
||||
from contextlib import asynccontextmanager
|
||||
|
||||
import httpx
|
||||
|
||||
from shonar.services.ai import ProviderConfigError, ProviderTransientError, SummaryResult
|
||||
from shonar.services.ai._llm import SYSTEM_PROMPT, build_user_message, parse_summary
|
||||
|
||||
|
||||
class OllamaProvider:
|
||||
name = "ollama"
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
base_url: str,
|
||||
model: str = "",
|
||||
timeout_s: float = 900.0,
|
||||
http_client: httpx.AsyncClient | None = None,
|
||||
) -> None:
|
||||
if not base_url.strip():
|
||||
raise ProviderConfigError("ollama needs SHONAR_LLM_BASE_URL.")
|
||||
if not model.strip():
|
||||
raise ProviderConfigError("ollama needs SHONAR_LLM_MODEL.")
|
||||
self.base_url = base_url.rstrip("/")
|
||||
self.model = model
|
||||
self.timeout_s = timeout_s
|
||||
self.http_client = http_client
|
||||
|
||||
@asynccontextmanager
|
||||
async def _client(self) -> AsyncIterator[httpx.AsyncClient]:
|
||||
if self.http_client is not None:
|
||||
yield self.http_client
|
||||
else:
|
||||
async with httpx.AsyncClient(timeout=self.timeout_s) as client:
|
||||
yield client
|
||||
|
||||
async def summarize(self, transcript: str, *, title: str | None = None) -> SummaryResult:
|
||||
payload = {
|
||||
"model": self.model,
|
||||
"stream": False,
|
||||
"format": "json",
|
||||
# qwen3-family models "think" by default: a long reasoning
|
||||
# chain before the JSON answer, brutally slow on CPU and it
|
||||
# does not improve the summary. Ask for the answer directly
|
||||
# (ignored by non-thinking models).
|
||||
"think": False,
|
||||
"options": {"num_ctx": 8192},
|
||||
"messages": [
|
||||
{"role": "system", "content": SYSTEM_PROMPT},
|
||||
{"role": "user", "content": build_user_message(transcript, title)},
|
||||
],
|
||||
}
|
||||
try:
|
||||
async with self._client() as client:
|
||||
resp = await client.post(f"{self.base_url}/api/chat", json=payload)
|
||||
except (httpx.TimeoutException, httpx.TransportError) as e:
|
||||
raise ProviderTransientError(f"Ollama unreachable: {type(e).__name__}") from e
|
||||
if resp.status_code == 404:
|
||||
# Missing model and missing route both 404 here; both are
|
||||
# configuration, not weather.
|
||||
raise ProviderConfigError("Ollama has no such model or route (HTTP 404).")
|
||||
if resp.status_code != 200:
|
||||
raise ProviderTransientError(f"Summarization failed (HTTP {resp.status_code}).")
|
||||
try:
|
||||
content = resp.json()["message"]["content"]
|
||||
except (ValueError, KeyError, TypeError) as e:
|
||||
raise ProviderTransientError("Ollama sent an unreadable reply.") from e
|
||||
import json as _json
|
||||
|
||||
try:
|
||||
data = _json.loads(content)
|
||||
except ValueError as e:
|
||||
raise ProviderTransientError("Ollama reply was not JSON.") from e
|
||||
return parse_summary(data, self.model)
|
||||
Loading…
Add table
Add a link
Reference in a new issue