S.H.O.N.A.R._Desktop_Companion/backend/shonar/services/exports.py
avi 76c867fca4 Standalone Shonar Desktop: vendor portable sources + local engine; decouple from ~/Projects/Shonar
- shared/ = portable Android-origin sources vendored from deferred/desktop-server
  (app/build.gradle.kts srcDir repointed; PlaybackController.kt excluded as Android-only)
- backend/ = bundled-lite engine (SQLite + inline queue); .venv symlinked from the
  old checkout, PYTHONPATH pins THIS backend's code over any editable install
- repoRoot() resolves this project dir (env SHONAR_REPO still wins); desktop-dev.sh
  watches shared/ + backend/
- Verified: :app:compileKotlin + :app:test green (23 tests); engine boots on :8010,
  self-migrates, /healthz ok
2026-09-14 17:14:54 -05:00

251 lines
8.1 KiB
Python

"""Exports (M9): audio, transcript txt, notes markdown, bundle zip.
Synchronous generation — every artifact is small enough (text, or one
audio file) that a background job adds failure modes, not speed. Each
successful export records an ExportJob row and stores the produced bytes
as an Asset(kind=export) so the audit trail exists; the response is the
file itself (no separate download-asset round trip).
Formats:
audio — the original upload, byte-identical, original mime/extension
txt — current transcript text
md — notes.md: title, metadata, notes, summary sections, transcript
zip — bundle: original audio + transcript.txt + notes.md
"""
from __future__ import annotations
import io
import uuid
import zipfile
from dataclasses import dataclass
from sqlalchemy import select
from sqlalchemy.ext.asyncio import AsyncSession
from shonar.db.models import (
Asset,
AssetKind,
ExportJob,
JobStatus,
Recording,
Summary,
Transcript,
)
class ExportError(Exception):
def __init__(self, status_code: int, message: str):
super().__init__(message)
self.status_code = status_code
self.message = message
EXPORT_FORMATS = ("audio", "txt", "md", "zip")
@dataclass
class ExportResult:
filename: str
mime_type: str
data: bytes
def _safe_filename(rec: Recording) -> str:
"""Slug the title; fall back to the recorded_at stamp. Never leaks ids."""
base = "".join(
c if (c.isalnum() or c in "-_ ") else " " for c in (rec.title or "")
).strip()
if not base:
base = f"recording-{rec.recorded_at:%Y%m%d-%H%M%S}"
return base[:120]
async def _current_transcript(session: AsyncSession, rec_id: uuid.UUID) -> Transcript | None:
return await session.scalar(
select(Transcript)
.where(Transcript.recording_id == rec_id, Transcript.superseded_at.is_(None))
.order_by(Transcript.version.desc())
)
async def _current_summary(session: AsyncSession, rec_id: uuid.UUID) -> Summary | None:
return await session.scalar(
select(Summary)
.where(Summary.recording_id == rec_id, Summary.superseded_at.is_(None))
.order_by(Summary.version.desc())
)
async def _original_asset(session: AsyncSession, rec_id: uuid.UUID) -> Asset | None:
return await session.scalar(
select(Asset).where(Asset.recording_id == rec_id, Asset.kind == AssetKind.original)
)
def _summary_md(summary: Summary | None) -> str:
if summary is None:
return ""
c = summary.content or {}
out = ["## Summary\n"]
if c.get("short"):
out.append(f"{c['short']}\n")
if c.get("detailed"):
out.append(f"### Detailed\n\n{c['detailed']}\n")
for key, header in (
("key_points", "Key Points"),
("decisions", "Decisions"),
("action_items", "Action Items"),
("questions", "Questions"),
):
items = c.get(key)
if isinstance(items, list) and items:
out.append(f"### {header}\n")
out.extend(f"- {x}" for x in items)
out.append("")
return "\n".join(out)
def _transcript_md(t: Transcript | None) -> str:
if t is None:
return ""
lines = ["## Transcript\n"]
segs = [s for s in (t.segments or []) if isinstance(s, dict)]
if segs:
for s in segs:
start = float(s.get("start", 0.0))
stamp = f"{int(start // 60):02d}:{start % 60:04.1f}"
speaker = f"**{s['speaker']}**: " if s.get("speaker") else ""
lines.append(f"- `[{stamp}]` {speaker}{s.get('text', '').strip()}")
else:
lines.append(t.text or "")
return "\n".join(lines) + "\n"
def _notes_md(
rec: Recording, t: Transcript | None, summary: Summary | None,
tag_names: list[str] | None = None,
) -> str:
parts = [
f"# {rec.title}\n",
f"- Recorded: {rec.recorded_at:%Y-%m-%d %H:%M} UTC",
f"- Duration: {rec.duration_seconds:.1f}s",
]
if tag_names:
parts.append("- Tags: " + ", ".join(f"`{x}`" for x in tag_names))
parts.append("")
if rec.notes:
parts.append(f"## Notes\n\n{rec.notes}\n")
s = _summary_md(summary)
if s:
parts.append(s)
tr = _transcript_md(t)
if tr:
parts.append(tr)
return "\n".join(parts)
def _zip(files: list[tuple[str, bytes]]) -> bytes:
buf = io.BytesIO()
with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as z:
for name, data in files:
z.writestr(name, data)
return buf.getvalue()
async def build_export(
session: AsyncSession, rec: Recording, export_format: str
) -> ExportResult:
"""Build one export artifact for an owned, non-deleted recording."""
from shonar.storage import get_storage
if export_format not in EXPORT_FORMATS:
raise ExportError(422, f"Unknown export format. Use one of: {', '.join(EXPORT_FORMATS)}")
original = await _original_asset(session, rec.id)
t = await _current_transcript(session, rec.id)
summary = await _current_summary(session, rec.id)
from shonar.services.search import tag_names
tags = await tag_names(session, rec.id)
base = _safe_filename(rec)
if export_format == "audio":
if original is None:
raise ExportError(404, "No audio stored for this recording")
data = await get_storage().get(original.storage_key)
ext = original.storage_key[original.storage_key.rfind(".") :]
return ExportResult(filename=f"{base}{ext}", mime_type=original.mime_type, data=data)
if export_format == "txt":
if t is None:
raise ExportError(404, "No transcript yet — transcribe first")
return ExportResult(
filename=f"{base}.txt", mime_type="text/plain; charset=utf-8",
data=(t.text or "").encode("utf-8"),
)
if export_format == "md":
if t is None and summary is None and not rec.notes:
raise ExportError(
404, "Nothing to export — this recording has no notes, transcript, or summary"
)
return ExportResult(
filename=f"{base}.md", mime_type="text/markdown; charset=utf-8",
data=_notes_md(rec, t, summary, tags).encode("utf-8"),
)
# zip bundle: whatever exists, always at least the audio when present.
if original is None and t is None and summary is None and not rec.notes:
raise ExportError(404, "Nothing to export for this recording")
files: list[tuple[str, bytes]] = []
if original is not None:
audio = await get_storage().get(original.storage_key)
ext = original.storage_key[original.storage_key.rfind(".") :]
files.append((f"{base}{ext}", audio))
if t is not None:
files.append(("transcript.txt", (t.text or "").encode("utf-8")))
files.append(("notes.md", _notes_md(rec, t, summary, tags).encode("utf-8")))
return ExportResult(
filename=f"{base}.zip", mime_type="application/zip", data=_zip(files)
)
async def record_export(
session: AsyncSession, user_id: uuid.UUID, rec: Recording,
export_format: str, result: ExportResult,
) -> None:
"""Persist the audit trail: ExportJob(succeeded) + Asset(kind=export).
Best-effort storage of the artifact bytes; a storage failure never
fails the download the user already received.
"""
from shonar.storage import get_storage
job = ExportJob(
user_id=user_id,
recording_id=rec.id,
export_type=export_format,
status=JobStatus.succeeded,
)
session.add(job)
try:
key = f"exports/{user_id}/{rec.id}/{export_format}-{uuid.uuid4().hex}"
await get_storage().put(key, result.data)
import hashlib
asset = Asset(
recording_id=rec.id,
user_id=user_id,
kind=AssetKind.export,
storage_key=key,
mime_type=result.mime_type,
size_bytes=len(result.data),
checksum_sha256=hashlib.sha256(result.data).hexdigest(),
)
session.add(asset)
await session.flush()
job.asset_id = asset.id
except Exception: # noqa: BLE001 — audit copy is best-effort
pass
await session.flush()