Backend: Ollama CPU-friendly summarize (think=off, 900s timeout, 8k ctx)
qwen3-family models think by default: a long reasoning chain before the JSON answer that is brutally slow on CPU and doesn't improve summaries. Send think:false (ignored by non-thinking models), raise the request budget to 900s (first call loads a multi-GB model into RAM), and pin num_ctx=8192. Verified live: spoken standup clip transcribed by faster-whisper base and summarized by qwen3:4b through the inline queue in ~20s (warm), recording status completed.
This commit is contained in:
parent
1475f48329
commit
535a96a1f0
1 changed files with 7 additions and 1 deletions
|
|
@ -18,7 +18,7 @@ class OllamaProvider:
|
||||||
self,
|
self,
|
||||||
base_url: str,
|
base_url: str,
|
||||||
model: str = "",
|
model: str = "",
|
||||||
timeout_s: float = 300.0,
|
timeout_s: float = 900.0,
|
||||||
http_client: httpx.AsyncClient | None = None,
|
http_client: httpx.AsyncClient | None = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
if not base_url.strip():
|
if not base_url.strip():
|
||||||
|
|
@ -43,6 +43,12 @@ class OllamaProvider:
|
||||||
"model": self.model,
|
"model": self.model,
|
||||||
"stream": False,
|
"stream": False,
|
||||||
"format": "json",
|
"format": "json",
|
||||||
|
# qwen3-family models "think" by default: a long reasoning
|
||||||
|
# chain before the JSON answer, brutally slow on CPU and it
|
||||||
|
# does not improve the summary. Ask for the answer directly
|
||||||
|
# (ignored by non-thinking models).
|
||||||
|
"think": False,
|
||||||
|
"options": {"num_ctx": 8192},
|
||||||
"messages": [
|
"messages": [
|
||||||
{"role": "system", "content": SYSTEM_PROMPT},
|
{"role": "system", "content": SYSTEM_PROMPT},
|
||||||
{"role": "user", "content": build_user_message(transcript, title)},
|
{"role": "user", "content": build_user_message(transcript, title)},
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue