diff --git a/backend/shonar/services/ai/ollama.py b/backend/shonar/services/ai/ollama.py index 574f77a..dd293a5 100644 --- a/backend/shonar/services/ai/ollama.py +++ b/backend/shonar/services/ai/ollama.py @@ -18,7 +18,7 @@ class OllamaProvider: self, base_url: str, model: str = "", - timeout_s: float = 300.0, + timeout_s: float = 900.0, http_client: httpx.AsyncClient | None = None, ) -> None: if not base_url.strip(): @@ -43,6 +43,12 @@ class OllamaProvider: "model": self.model, "stream": False, "format": "json", + # qwen3-family models "think" by default: a long reasoning + # chain before the JSON answer, brutally slow on CPU and it + # does not improve the summary. Ask for the answer directly + # (ignored by non-thinking models). + "think": False, + "options": {"num_ctx": 8192}, "messages": [ {"role": "system", "content": SYSTEM_PROMPT}, {"role": "user", "content": build_user_message(transcript, title)},