From 535a96a1f02528b7736e27d36ecad115032eb940 Mon Sep 17 00:00:00 2001 From: avi Date: Sat, 12 Sep 2026 23:41:25 -0500 Subject: [PATCH] Backend: Ollama CPU-friendly summarize (think=off, 900s timeout, 8k ctx) qwen3-family models think by default: a long reasoning chain before the JSON answer that is brutally slow on CPU and doesn't improve summaries. Send think:false (ignored by non-thinking models), raise the request budget to 900s (first call loads a multi-GB model into RAM), and pin num_ctx=8192. Verified live: spoken standup clip transcribed by faster-whisper base and summarized by qwen3:4b through the inline queue in ~20s (warm), recording status completed. --- backend/shonar/services/ai/ollama.py | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/backend/shonar/services/ai/ollama.py b/backend/shonar/services/ai/ollama.py index 574f77a..dd293a5 100644 --- a/backend/shonar/services/ai/ollama.py +++ b/backend/shonar/services/ai/ollama.py @@ -18,7 +18,7 @@ class OllamaProvider: self, base_url: str, model: str = "", - timeout_s: float = 300.0, + timeout_s: float = 900.0, http_client: httpx.AsyncClient | None = None, ) -> None: if not base_url.strip(): @@ -43,6 +43,12 @@ class OllamaProvider: "model": self.model, "stream": False, "format": "json", + # qwen3-family models "think" by default: a long reasoning + # chain before the JSON answer, brutally slow on CPU and it + # does not improve the summary. Ask for the answer directly + # (ignored by non-thinking models). + "think": False, + "options": {"num_ctx": 8192}, "messages": [ {"role": "system", "content": SYSTEM_PROMPT}, {"role": "user", "content": build_user_message(transcript, title)},