S.H.O.N.A.R._Desktop_Companion/backend/shonar/services/retention.py
avi 76c867fca4 Standalone Shonar Desktop: vendor portable sources + local engine; decouple from ~/Projects/Shonar
- shared/ = portable Android-origin sources vendored from deferred/desktop-server
  (app/build.gradle.kts srcDir repointed; PlaybackController.kt excluded as Android-only)
- backend/ = bundled-lite engine (SQLite + inline queue); .venv symlinked from the
  old checkout, PYTHONPATH pins THIS backend's code over any editable install
- repoRoot() resolves this project dir (env SHONAR_REPO still wins); desktop-dev.sh
  watches shared/ + backend/
- Verified: :app:compileKotlin + :app:test green (23 tests); engine boots on :8010,
  self-migrates, /healthz ok
2026-09-14 17:14:54 -05:00

101 lines
3.7 KiB
Python

"""Retention sweep (M9): hard-delete what passed its grace window.
Two policies, both driven by ``deleted_at``:
* **Recordings** soft-deleted more than ``retention_grace_days`` ago have
their rows removed (cascade cleans transcripts/summaries/jobs/tags) and
every stored asset file deleted best-effort.
* **Accounts** deleted more than ``retention_grace_days`` ago are hard
deleted (user cascade takes their recordings/assets/devices/tokens);
their storage files are collected the same way.
Running inside the grace window is a no-op, so an accidental delete stays
cancellable until the sweep actually fires. The sweep is idempotent and
safe to run on any cadence (worker cron + inline-queue timer).
"""
from __future__ import annotations
import contextlib
import logging
from datetime import timedelta
from sqlalchemy import delete as sql_delete
from sqlalchemy import select
from shonar.core.config import get_settings
from shonar.db.models import Asset, Recording, User, utcnow
from shonar.db.session import session_factory
from shonar.storage import get_storage
logger = logging.getLogger("shonar.retention")
async def sweep_deleted(limit: int = 200) -> dict[str, int]:
"""Hard-purge expired recordings and accounts. Returns counts."""
grace = timedelta(days=get_settings().retention_grace_days)
cutoff = utcnow() - grace
purged = {"recordings": 0, "users": 0, "files": 0}
storage = get_storage()
async with session_factory()() as session:
# --- recordings (skip rows whose account is also expiring: the
# user cascade below collects their files in one pass) ---
expiring_users = select(User.id).where(
User.deleted_at.is_not(None), User.deleted_at < cutoff
)
recs = list(
await session.scalars(
select(Recording)
.where(
Recording.deleted_at.is_not(None),
Recording.deleted_at < cutoff,
Recording.user_id.notin_(expiring_users),
)
.limit(limit)
)
)
for rec in recs:
assets = list(
await session.scalars(select(Asset).where(Asset.recording_id == rec.id))
)
await session.delete(rec)
await session.flush()
for a in assets:
with contextlib.suppress(Exception): # best effort; DB row is gone
await storage.delete(a.storage_key)
purged["files"] += 1
purged["recordings"] += 1
# --- accounts ---
users = list(
await session.scalars(
select(User)
.where(User.deleted_at.is_not(None), User.deleted_at < cutoff)
.limit(limit)
)
)
for user in users:
assets = list(
await session.scalars(select(Asset).where(Asset.user_id == user.id))
)
keys = [a.storage_key for a in assets]
# DB-level delete: the ORM would null the NOT NULL FKs of the
# user's recordings before the ON DELETE CASCADE could fire.
await session.execute(sql_delete(User).where(User.id == user.id))
await session.flush()
for key in keys:
with contextlib.suppress(Exception):
await storage.delete(key)
purged["files"] += 1
purged["users"] += 1
await session.commit()
if purged["recordings"] or purged["users"]:
logger.info(
"retention sweep purged %d recordings, %d accounts, %d files (grace %dd)",
purged["recordings"], purged["users"], purged["files"],
get_settings().retention_grace_days,
)
return purged