S.H.O.N.A.R._Desktop_Companion/backend/shonar/services/ai/whisper_http.py
avi 76c867fca4 Standalone Shonar Desktop: vendor portable sources + local engine; decouple from ~/Projects/Shonar
- shared/ = portable Android-origin sources vendored from deferred/desktop-server
  (app/build.gradle.kts srcDir repointed; PlaybackController.kt excluded as Android-only)
- backend/ = bundled-lite engine (SQLite + inline queue); .venv symlinked from the
  old checkout, PYTHONPATH pins THIS backend's code over any editable install
- repoRoot() resolves this project dir (env SHONAR_REPO still wins); desktop-dev.sh
  watches shared/ + backend/
- Verified: :app:compileKotlin + :app:test green (23 tests); engine boots on :8010,
  self-migrates, /healthz ok
2026-09-14 17:14:54 -05:00

128 lines
4.3 KiB
Python

"""Transcription via any OpenAI-compatible `/v1/audio/transcriptions`
endpoint (self-hosted whisper.cpp server, commercial Whisper API, …).
Sends `verbose_json` so segment timings come back with the text.
"""
from __future__ import annotations
from collections.abc import AsyncIterator
from contextlib import asynccontextmanager
import httpx
from shonar.services.ai import (
ProviderConfigError,
ProviderTransientError,
Segment,
TranscriptResult,
)
class WhisperHttpProvider:
name = "whisper_http"
def __init__(
self,
base_url: str,
model: str = "base",
api_key: str = "",
timeout_s: float = 300.0,
http_client: httpx.AsyncClient | None = None,
) -> None:
if not base_url.strip():
raise ProviderConfigError(
"whisper_http needs SHONAR_TRANSCRIPTION_BASE_URL."
)
self.base_url = base_url.rstrip("/")
self.model = model
self.api_key = api_key
self.timeout_s = timeout_s
self.http_client = http_client
@asynccontextmanager
async def _client(self) -> AsyncIterator[httpx.AsyncClient]:
if self.http_client is not None:
yield self.http_client
else:
async with httpx.AsyncClient(timeout=self.timeout_s) as client:
yield client
async def transcribe(
self,
audio: bytes,
mime: str,
*,
language_hint: str | None = None,
on_progress=None, # accepted for protocol parity; not reported
) -> TranscriptResult:
headers = (
{"Authorization": f"Bearer {self.api_key}"} if self.api_key else {}
)
data: dict[str, str] = {"model": self.model, "response_format": "verbose_json"}
if language_hint:
data["language"] = language_hint
files = {"file": (f"audio.{_ext(mime)}", audio, mime or "application/octet-stream")}
try:
async with self._client() as client:
resp = await client.post(
f"{self.base_url}/v1/audio/transcriptions",
headers=headers,
data=data,
files=files,
)
except (httpx.TimeoutException, httpx.TransportError) as e:
raise ProviderTransientError(
f"Transcription service unreachable: {type(e).__name__}"
) from e
if resp.status_code in (401, 403, 404):
raise ProviderConfigError(
f"Transcription service refused the request (HTTP {resp.status_code})."
)
if resp.status_code == 429 or resp.status_code >= 500:
raise ProviderTransientError(
f"Transcription service busy (HTTP {resp.status_code})."
)
if resp.status_code != 200:
raise ProviderTransientError(
f"Transcription failed (HTTP {resp.status_code})."
)
try:
body = resp.json()
except ValueError as e:
raise ProviderTransientError("Transcription service sent no JSON.") from e
segments = []
raw_segs = body.get("segments")
if isinstance(raw_segs, list):
for s in raw_segs:
if not isinstance(s, dict):
continue
try:
segments.append(
Segment(
start=float(s.get("start", 0.0)),
end=float(s.get("end", 0.0)),
text=str(s.get("text", "")),
)
)
except (TypeError, ValueError):
continue
text = body.get("text")
return TranscriptResult(
text=text if isinstance(text, str) else "",
language=body.get("language") if isinstance(body.get("language"), str) else None,
segments=segments,
model=self.model,
)
def _ext(mime: str) -> str:
return {
"audio/mp4": "m4a",
"audio/m4a": "m4a",
"audio/wav": "wav",
"audio/x-wav": "wav",
"audio/ogg": "ogg",
"audio/opus": "ogg",
"audio/webm": "webm",
"audio/mpeg": "mp3",
}.get(mime.lower().split(";")[0].strip(), "bin")