fix(cassettes): order state events by created_at, not by one remembered id

The gate on the ATM-state consumer compared the incoming event id against
the id stored on a single arbitrary row (SELECT ... LIMIT 1, no ORDER BY).
That is a one-event memory, not a watermark: a re-delivered A, B, A applied
three times. Worse, created_at was parsed, written to state_at and then
never compared, so an event arriving late overwrote newer state — nothing
in the path ever looked at the clock.

Events are now applied only when strictly newer than the OLDEST state stamp
on file. Strict '>' subsumes replay dedup, since a replay carries the same
stamp. Oldest rather than newest is deliberate: every execute in this data
layer commits on its own, so a multi-row apply cannot be made atomic here,
and gating on the oldest means a crash mid-apply is re-applied on the next
event instead of being mistaken for a complete one. The ATM republishes on
a heartbeat, so it converges.

Stamps are compared as unix floats because SQLite returns integers,
Postgres returns timestamps and the incoming value is tz-aware; comparing
raw would either raise or quietly mislead. An unparseable incoming stamp
fails closed.

Also renames apply_bootstrap_state to apply_reported_state and corrects the
module comments. There has never been a once-per-machine guard, so calling
it a one-shot bootstrap consumer described something the code did not do.

Refs #43

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
Padreug 2026-09-22 22:23:19 +02:00
commit 27449e1d11
4 changed files with 144 additions and 85 deletions

108
crud.py
View file

@ -5,7 +5,7 @@
# through dca_machines.operator_user_id. See plan section "Identity & multi-
# machine model".
from datetime import datetime
from datetime import datetime, timezone
from lnbits.db import Database
from lnbits.helpers import urlsafe_short_hash
@ -1442,7 +1442,7 @@ async def upsert_fleet_snapshot(
# Cassette configs — operator-driven ATM cassette inventory (#29 v1.1).
# =============================================================================
# Row lifecycle per #29:
# - First population for a (machine_id, position) pair → apply_bootstrap_state
# - First population for a (machine_id, position) pair → apply_reported_state
# (consumer reading the ATM's one-shot bitspire-cassettes-state event)
# - Operator edit of denomination or count → update_cassette_config
# (refuses to create new rows; the slot count is hardware-determined)
@ -1450,25 +1450,56 @@ async def upsert_fleet_snapshot(
# re-provisioning + new bootstrap event (not exposed in v1 here)
def _should_apply_bootstrap_state(
existing_state_event_id: str | None, incoming_event_id: str
) -> bool:
"""Pure-function dedup gate for apply_bootstrap_state.
def _as_unix(value) -> float | None:
"""Normalise whatever the driver hands back for state_at to a unix float.
Returns False if any existing row for this machine already references
the incoming event_id (relay re-delivery after restart). True otherwise.
Extracted as a pure function so the dedup decision is unit-testable
without a database round-trip. The actual idempotency check in
apply_bootstrap_state fetches one existing row and passes its
state_event_id here.
SQLite stores these as integers and Postgres as timestamps, and an event's
created_at arrives tz-aware, so comparing the raw values risks either a
TypeError (aware vs naive) or a silently wrong answer. Everything is
compared as seconds since the epoch instead.
"""
return existing_state_event_id != incoming_event_id
if value is None:
return None
if isinstance(value, datetime):
if value.tzinfo is None:
value = value.replace(tzinfo=timezone.utc)
return value.timestamp()
try:
return float(value)
except (TypeError, ValueError):
return None
async def get_cassette_config(
machine_id: str, position: int
) -> CassetteConfig | None:
def _should_apply_state_event(oldest_state_at, incoming_created_at) -> bool:
"""Ordering gate for apply_reported_state.
Applies only when the incoming event is strictly newer than the OLDEST
state stamp on file for the machine.
Two deliberate choices:
- Compare created_at, not event ids. The old gate asked only whether the
incoming id differed from one stored row, which is a one-event memory:
a re-delivered A, B, A applied three times, and an event that arrived
late overwrote newer state because nothing ever looked at the clock.
Strict `>` also subsumes exact-replay dedup, since a replay carries the
same stamp.
- Oldest, not newest. Every execute in this data layer commits on its own,
so a multi-row apply cannot be made atomic here; a crash mid-loop leaves
some rows advanced and some not. Gating on the oldest means a partial
apply is re-applied on the next event rather than being mistaken for a
complete one, and the ATM republishes on a heartbeat, so it converges.
"""
oldest = _as_unix(oldest_state_at)
if oldest is None:
return True
incoming = _as_unix(incoming_created_at)
if incoming is None:
return False
return incoming > oldest
async def get_cassette_config(machine_id: str, position: int) -> CassetteConfig | None:
return await db.fetchone(
"SELECT * FROM spirekeeper.cassette_configs "
"WHERE machine_id = :mid AND position = :pos",
@ -1497,7 +1528,7 @@ async def update_cassette_config(
) -> CassetteConfig | None:
"""Operator-driven row update: change denomination and/or count for a
single cassette slot. Refuses to create new rows — those only land via
apply_bootstrap_state() consuming an ATM bootstrap event (per #29 row
apply_reported_state() consuming an ATM bootstrap event (per #29 row
lifecycle: hardware-determined slot count, not operator-creatable).
Returns None if the (machine_id, position) row doesn't exist.
"""
@ -1520,41 +1551,40 @@ async def update_cassette_config(
return await get_cassette_config(machine_id, position)
async def apply_bootstrap_state(
async def apply_reported_state(
machine_id: str,
event_id: str,
event_created_at: datetime,
payload: PublishCassettesPayload,
) -> bool:
"""Consume an ATM-published kind-30078 bitspire-cassettes-state:<m> event
and upsert one cassette_configs row per position in the payload.
and reconcile cassette_configs for the machine against it.
Returns True if the upsert ran; False if any existing row for this
machine already references this event_id (idempotent on relay
re-delivery / restart).
Returns True if the state was applied, False if the event was not newer
than what is already on file (see _should_apply_state_event).
Populates both the operator-believed columns (denomination, count,
updated_at, updated_by='atm-bootstrap') AND the v2 reverse-channel
columns (state_denomination, state_count, state_at, state_event_id)
so the operator's initial view matches the ATM's reported state. v2
reconciliation UI will diverge them when continuous reverse-channel
events land + the operator subsequently edits.
updated_at, updated_by) and the reported columns (state_denomination,
state_count, state_at, state_event_id), so the UI can show reported
against believed.
"""
existing_first: dict | None = await db.fetchone(
"SELECT state_event_id FROM spirekeeper.cassette_configs "
"WHERE machine_id = :mid LIMIT 1",
oldest: dict | None = await db.fetchone(
"SELECT state_at FROM spirekeeper.cassette_configs "
"WHERE machine_id = :mid AND state_at IS NOT NULL "
"ORDER BY state_at ASC LIMIT 1",
{"mid": machine_id},
)
existing_event_id: str | None = None
if existing_first is not None:
existing_event_id = (
existing_first.get("state_event_id")
if isinstance(existing_first, dict)
else getattr(existing_first, "state_event_id", None)
oldest_state_at = None
if oldest is not None:
oldest_state_at = (
oldest.get("state_at")
if isinstance(oldest, dict)
else getattr(oldest, "state_at", None)
)
if not _should_apply_bootstrap_state(existing_event_id, event_id):
if not _should_apply_state_event(oldest_state_at, event_created_at):
return False
now = datetime.now()
for pos, row in payload.positions.items():
await db.execute(
@ -1581,7 +1611,7 @@ async def apply_bootstrap_state(
"denom": row.denomination,
"count": row.count,
"now": now,
"by": "atm-bootstrap",
"by": "atm-report",
"state_denom": row.denomination,
"state_count": row.count,
"state_at": event_created_at,