From 44f5c0dbcf4ac4b9edde895cbcb646ee6c504b9c Mon Sep 17 00:00:00 2001 From: Patrick Mulligan Date: Tue, 4 Aug 2026 18:11:40 +0200 Subject: [PATCH 1/3] =?UTF-8?q?docs(adr):=20amend=20ADR-002=20=E2=80=94=20?= =?UTF-8?q?app=20recovery=20is=20first-line,=20SSH/NetBird=20last-resort?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The access/recovery plane (SSH/NetBird) stands and may carry recovery procedures, but it is explicitly NOT the only or first-line recovery. Add a layered, cheapest-first recovery model: (1) app auto-recovery of its own relay/Lightning connectivity, (2) an on-screen Retry for an operator at the kiosk, (3) SSH/NetBird as the last-resort remote plane for genuine app/OS failure. A public kiosk must not need remote shell access to recover from a transient/boot-before-network outage. Co-Authored-By: Claude Opus 4.8 --- .../002-remote-access-and-fleet-management.md | 30 +++++++++++++++++++ 1 file changed, 30 insertions(+) diff --git a/docs/adr/002-remote-access-and-fleet-management.md b/docs/adr/002-remote-access-and-fleet-management.md index 4931daa..a518930 100644 --- a/docs/adr/002-remote-access-and-fleet-management.md +++ b/docs/adr/002-remote-access-and-fleet-management.md @@ -104,6 +104,36 @@ The only honest way for a machine operator to exclude the SaaS operator is to ** - The machine operator's Nostr key can become the single root of trust across all three planes — SSH `authorized_keys` + VPN enrollment at install, `AddOperator`/`RevokeOperator` (#42) for delegation — so granting/revoking any party (including the SaaS operator) is one scoped, revocable capability model. - `sshd` posture should be tightened to key-only for deployed boxes (password auth is currently forced on for installed configs for first-boot provisioning; scope it to the LAN/first-boot window). Tracks with [#51](https://git.atitlan.io/aiolabs/bitspire/issues/51). +## Amendment (2026-08-04): the access/recovery plane is not the *only* recovery + +**Status:** Accepted · **Context:** the ATM app had no way to recover its own +connectivity — a machine that booted with no internet (or whose init otherwise +failed) sat on "ATM Unavailable" until a manual `systemctl restart bitspire`, +even after the network came back. + +This ADR's SSH/NetBird recovery plane stands — it is the operator's +**app-and-OS-independent** path for the unanticipated and the broken, and may +carry recovery *procedures* (restart the service, inspect logs, re-provision). +But it is explicitly **not the first-line and not the only recovery method.** +Recovery is layered, cheapest-first: + +1. **App auto-recovery (first-line, no human).** The ATM app recovers its own + relay/Lightning connectivity when possible: the nostr client already + reconnects with backoff, and the app now re-initializes when connectivity + returns (a fresh renderer reload — HAL is preserved in the main process), + so "internet came back" self-heals without anyone touching the machine. +2. **On-screen manual retry (operator at the machine).** The maintenance + ("ATM Unavailable") screen carries a **Retry** button so a person standing + at the kiosk can force an immediate recovery attempt without shell access. +3. **SSH/NetBird (operator remote, last resort).** This plane — for when the + app *can't* self-heal or the box is genuinely broken. Unchanged by this + amendment beyond the reframing: it is the floor, not the front line. + +Rationale: the common failure (transient network / boot-before-network) must +not require remote shell access to a public kiosk. Reserve the heavyweight +recovery plane for genuine app/OS failure. Implemented on branch +`feat/connection-recovery`. + ## References - [#41](https://git.atitlan.io/aiolabs/bitspire/issues/41) — Multi-location deployment: runtime site config (the access plane's per-machine identity is provisioned here, not baked into the closure). From 6a833d357e1675bd6d933afa3125d0dd82e007ba Mon Sep 17 00:00:00 2001 From: Patrick Mulligan Date: Tue, 4 Aug 2026 18:11:40 +0200 Subject: [PATCH 2/3] feat(machine): connectivity auto-recovery + on-screen Retry MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A connectivity-type init failure (e.g. "No connected relays" when the box boots before the network) landed on "ATM Unavailable" permanently: init is one-shot and the nostr reconnect only helps after a first successful connect, so a machine never self-healed when internet returned. Recover by reloading the renderer, which re-runs init from a clean JS context (no leaked actors/subscriptions) while the main process keeps HAL: - main.ts: new `app:recover` IPC → reloadRenderer() (resets secretsConsumed). - hal:init is now idempotent (reuse the existing instance) so the reload — and the pre-existing watchdog crash-reload — can't double-open serial ports. - App.vue: when initError is a connectivity type (not the operator/ self-clearing states unpaired/awaiting-fees/maintenance), watch for the `online` event (recover immediately) plus a 45s backoff safety net, and render a kiosk-sized Retry button for a person at the machine. Preserves pairing + /var/lib state (renderer reload, not a process restart). Co-Authored-By: Claude Opus 4.8 --- apps/machine/electron/main.ts | 22 ++++++++ apps/machine/electron/preload.ts | 3 ++ apps/machine/src/App.vue | 78 +++++++++++++++++++++++++++- apps/machine/src/types/electron.d.ts | 2 + 4 files changed, 104 insertions(+), 1 deletion(-) diff --git a/apps/machine/electron/main.ts b/apps/machine/electron/main.ts index fc48d87..ac31a6a 100644 --- a/apps/machine/electron/main.ts +++ b/apps/machine/electron/main.ts @@ -404,6 +404,17 @@ ipcMain.handle('app:relaunch', (): void => { app.exit(0) }) +// Connectivity recovery: reload the renderer to re-run init from a clean slate +// (fresh JS context → no leaked actors/subscriptions), while preserving HAL in +// this main process (reloadRenderer resets secretsConsumed so get-atm-secrets +// works again, and hal:init is idempotent). The renderer calls this when it's +// stuck on a connectivity-type "ATM Unavailable" and the network returns, or +// when the operator taps the on-screen Retry (ADR-002 amendment 2026-08-04). +ipcMain.handle('app:recover', (): void => { + console.log('[Recovery] Reloading renderer to re-attempt initialization') + reloadRenderer() +}) + // State persistence IPC handlers ipcMain.handle('state:load-cassettes', () => loadCassettes()) ipcMain.handle('state:set-cassettes', (_event, cassettes) => setCassettes(cassettes)) @@ -480,6 +491,17 @@ let pendingBillDenomination: number | null = null ipcMain.handle('hal:init', async (_event, config) => { try { + // Idempotent: HAL lives in this (long-lived) main process, but the renderer + // re-runs full init on every reload — the watchdog's crash-recovery reload + // and the connectivity-recovery reload (app:recover) both re-invoke this. + // initializeHal opens serial ports without closing prior handles, so + // re-entering it would double-open the validator/dispenser. Reuse the + // existing instance instead; its validator event wiring already targets the + // (reloaded) mainWindow, so the reloaded renderer keeps receiving bill events. + if (halInstance) { + console.log('[Electron] HAL already initialized — reusing existing instance') + return { success: true } + } // Override cassette config with DB values (operator may have changed them via atm-tui // or via an operator-config publish from satmachineadmin). Pass per-position so the // HAL knows about every bay including duplicates of the same denomination — real diff --git a/apps/machine/electron/preload.ts b/apps/machine/electron/preload.ts index a811eab..41c54df 100644 --- a/apps/machine/electron/preload.ts +++ b/apps/machine/electron/preload.ts @@ -124,6 +124,8 @@ contextBridge.exposeInMainWorld('electronAPI', { // then relaunch so the normal boot flow pairs it. saveSpireSeed: (seed: string): Promise => ipcRenderer.invoke('state:save-spire-seed', seed), relaunchApp: (): Promise => ipcRenderer.invoke('app:relaunch'), + // Reload the renderer to re-attempt initialization (connectivity recovery). + recoverApp: (): Promise => ipcRenderer.invoke('app:recover'), applyOperatorCassettesConfig: ( payload: { @@ -243,6 +245,7 @@ declare global { resetForRepair: () => Promise saveSpireSeed: (seed: string) => Promise relaunchApp: () => Promise + recoverApp: () => Promise applyOperatorCassettesConfig: ( payload: { positions: Record }, eventCreatedAt: number diff --git a/apps/machine/src/App.vue b/apps/machine/src/App.vue index 2b89eb7..d9bce92 100644 --- a/apps/machine/src/App.vue +++ b/apps/machine/src/App.vue @@ -1,5 +1,5 @@