feat(machine): connectivity auto-recovery + on-screen Retry #82

Merged
padreug merged 3 commits from feat/connection-recovery into dev 2026-08-05 02:33:01 +00:00
6 changed files with 169 additions and 34 deletions

View file

@ -404,6 +404,17 @@ ipcMain.handle('app:relaunch', (): void => {
app.exit(0)
})
// Connectivity recovery: reload the renderer to re-run init from a clean slate
// (fresh JS context → no leaked actors/subscriptions), while preserving HAL in
// this main process (reloadRenderer resets secretsConsumed so get-atm-secrets
// works again, and hal:init is idempotent). The renderer calls this when it's
// stuck on a connectivity-type "ATM Unavailable" and the network returns, or
// when the operator taps the on-screen Retry (ADR-002 amendment 2026-08-04).
ipcMain.handle('app:recover', (): void => {
console.log('[Recovery] Reloading renderer to re-attempt initialization')
reloadRenderer()
})
// State persistence IPC handlers
ipcMain.handle('state:load-cassettes', () => loadCassettes())
ipcMain.handle('state:set-cassettes', (_event, cassettes) => setCassettes(cassettes))
@ -480,6 +491,17 @@ let pendingBillDenomination: number | null = null
ipcMain.handle('hal:init', async (_event, config) => {
try {
// Idempotent: HAL lives in this (long-lived) main process, but the renderer
// re-runs full init on every reload — the watchdog's crash-recovery reload
// and the connectivity-recovery reload (app:recover) both re-invoke this.
// initializeHal opens serial ports without closing prior handles, so
// re-entering it would double-open the validator/dispenser. Reuse the
// existing instance instead; its validator event wiring already targets the
// (reloaded) mainWindow, so the reloaded renderer keeps receiving bill events.
if (halInstance) {
console.log('[Electron] HAL already initialized — reusing existing instance')
return { success: true }
}
// Override cassette config with DB values (operator may have changed them via atm-tui
// or via an operator-config publish from satmachineadmin). Pass per-position so the
// HAL knows about every bay including duplicates of the same denomination — real

View file

@ -124,6 +124,8 @@ contextBridge.exposeInMainWorld('electronAPI', {
// then relaunch so the normal boot flow pairs it.
saveSpireSeed: (seed: string): Promise<void> => ipcRenderer.invoke('state:save-spire-seed', seed),
relaunchApp: (): Promise<void> => ipcRenderer.invoke('app:relaunch'),
// Reload the renderer to re-attempt initialization (connectivity recovery).
recoverApp: (): Promise<void> => ipcRenderer.invoke('app:recover'),
applyOperatorCassettesConfig: (
payload: {
@ -243,6 +245,7 @@ declare global {
resetForRepair: () => Promise<void>
saveSpireSeed: (seed: string) => Promise<void>
relaunchApp: () => Promise<void>
recoverApp: () => Promise<void>
applyOperatorCassettesConfig: (
payload: { positions: Record<string, { denomination: number; count: number }> },
eventCreatedAt: number

View file

@ -1,5 +1,5 @@
<script setup lang="ts">
import { onMounted, onUnmounted, ref, computed } from 'vue'
import { onMounted, onUnmounted, ref, computed, watch } from 'vue'
import { useRoute } from 'vue-router'
import { useAtmStore } from '@/stores/atm'
import { useTheme } from '@/composables/useTheme'
@ -158,8 +158,71 @@ onMounted(async () => {
onUnmounted(() => {
atmStore.stopPricePolling()
stopRecoveryWatch()
})
// ── Connectivity recovery (ADR-002 amendment 2026-08-04) ──────────────────
// A connectivity-type init failure lands on "ATM Unavailable" and, without
// this, stays there forever (init is one-shot; the nostr reconnect only helps
// AFTER a first successful connect). We recover by reloading the renderer —
// which re-runs this whole init from a clean JS context while the main process
// keeps HAL (see main.ts app:recover). Not for the operator/self-clearing
// states: `unpaired` shows the pairing wizard, `awaiting-fees` clears itself on
// the operator's fee-config event, `maintenance` is operator-set.
const NON_RECOVERABLE = new Set(['maintenance', 'awaiting-fees', 'unpaired'])
const isRecoverable = computed(
() => !!atmStore.initError && !NON_RECOVERABLE.has(atmStore.initError)
)
const recovering = ref(false)
const RECOVERY_RETRY_MS = 45_000
let recoveryTimer: ReturnType<typeof setInterval> | null = null
function triggerRecovery() {
if (recovering.value) return
recovering.value = true
console.log('[App] Attempting connectivity recovery (renderer reload)')
if (window.electronAPI?.recoverApp) {
void window.electronAPI.recoverApp() // main reloads renderer → fresh init
} else {
location.reload() // browser-dev fallback
}
}
function onOnline() {
// Network came back — recover immediately rather than waiting for the timer.
triggerRecovery()
}
function startRecoveryWatch() {
stopRecoveryWatch()
window.addEventListener('online', onOnline)
// Safety net for the online-but-relay-unreachable case (navigator.onLine only
// reflects a local route, not relay reachability).
recoveryTimer = setInterval(triggerRecovery, RECOVERY_RETRY_MS)
}
function stopRecoveryWatch() {
window.removeEventListener('online', onOnline)
if (recoveryTimer !== null) {
clearInterval(recoveryTimer)
recoveryTimer = null
}
}
/** Operator-facing "Retry" button on the maintenance screen. */
function retryNow() {
triggerRecovery()
}
watch(
isRecoverable,
(recoverable) => {
if (recoverable) startRecoveryWatch()
else stopRecoveryWatch()
},
{ immediate: true }
)
function toggleLiveServices() {
if (atmStore.useLiveServices) {
// Switch to mock
@ -211,6 +274,19 @@ function toggleLiveServices() {
>
{{ atmStore.initError }}
</p>
<!-- Manual recovery for a connectivity failure; auto-recovery also runs
in the background (online event + backoff). Not shown for operator/
self-clearing states (maintenance / awaiting-fees / unpaired). -->
<Button
v-if="isRecoverable"
size="kiosk"
:disabled="recovering"
class="mt-4"
@click="retryNow"
>
{{ recovering ? 'Retrying…' : 'Retry' }}
</Button>
</div>
<template v-else>

View file

@ -101,6 +101,8 @@ declare global {
resetForRepair: () => Promise<void>
saveSpireSeed: (seed: string) => Promise<void>
relaunchApp: () => Promise<void>
/** Reload the renderer to re-attempt initialization (connectivity recovery). */
recoverApp: () => Promise<void>
applyOperatorCassettesConfig: (
payload: { positions: Record<string, { denomination: number; count: number }> },
eventCreatedAt: number

View file

@ -104,6 +104,36 @@ The only honest way for a machine operator to exclude the SaaS operator is to **
- The machine operator's Nostr key can become the single root of trust across all three planes — SSH `authorized_keys` + VPN enrollment at install, `AddOperator`/`RevokeOperator` (#42) for delegation — so granting/revoking any party (including the SaaS operator) is one scoped, revocable capability model.
- `sshd` posture should be tightened to key-only for deployed boxes (password auth is currently forced on for installed configs for first-boot provisioning; scope it to the LAN/first-boot window). Tracks with [#51](https://git.atitlan.io/aiolabs/bitspire/issues/51).
## Amendment (2026-08-04): the access/recovery plane is not the *only* recovery
**Status:** Accepted · **Context:** the ATM app had no way to recover its own
connectivity — a machine that booted with no internet (or whose init otherwise
failed) sat on "ATM Unavailable" until a manual `systemctl restart bitspire`,
even after the network came back.
This ADR's SSH/NetBird recovery plane stands — it is the operator's
**app-and-OS-independent** path for the unanticipated and the broken, and may
carry recovery *procedures* (restart the service, inspect logs, re-provision).
But it is explicitly **not the first-line and not the only recovery method.**
Recovery is layered, cheapest-first:
1. **App auto-recovery (first-line, no human).** The ATM app recovers its own
relay/Lightning connectivity when possible: the nostr client already
reconnects with backoff, and the app now re-initializes when connectivity
returns (a fresh renderer reload — HAL is preserved in the main process),
so "internet came back" self-heals without anyone touching the machine.
2. **On-screen manual retry (operator at the machine).** The maintenance
("ATM Unavailable") screen carries a **Retry** button so a person standing
at the kiosk can force an immediate recovery attempt without shell access.
3. **SSH/NetBird (operator remote, last resort).** This plane — for when the
app *can't* self-heal or the box is genuinely broken. Unchanged by this
amendment beyond the reframing: it is the floor, not the front line.
Rationale: the common failure (transient network / boot-before-network) must
not require remote shell access to a public kiosk. Reserve the heavyweight
recovery plane for genuine app/OS failure. Implemented on branch
`feat/connection-recovery`.
## References
- [#41](https://git.atitlan.io/aiolabs/bitspire/issues/41) — Multi-location deployment: runtime site config (the access plane's per-machine identity is provisioned here, not baked into the closure).

View file

@ -292,6 +292,37 @@
# at ttyJ5, dispenser at ttyJ7 layout) — reuse the same hw module.
sintra-installed = mkInstalledConfig "sintra" ./deploy/nixos/hardware/upboard.nix;
batm3-installed = mkInstalledConfig "batm3" ./deploy/nixos/hardware/batm3.nix;
# USB-bootable variant of batm3-installed. This is the config the
# flashed USB stick actually runs — distinct fs labels so stage-1 can't
# latch the internal drive, nofail /boot, no growPartition, autoUpgrade
# off. Exposed as a named config (not just inline in the disk-image
# target) so its system closure can be built here and deployed in-place
# with `nix copy` + `switch-to-configuration` — updating the app on a
# running stick WITHOUT reflashing (preserves pairing + /var/lib state).
# disk-image-batm3-usb builds its filesystem image from this same config.
batm3-usb = self.nixosConfigurations.batm3-installed.extendModules {
modules = [
({ lib, ... }: {
fileSystems."/".device = lib.mkForce "/dev/disk/by-label/nixos-usb";
fileSystems."/boot".device = lib.mkForce "/dev/disk/by-label/ESP-USB";
# /boot must NOT be a hard boot dependency on the USB image. The
# firmware already loaded the bootloader before Linux; without
# nofail, a slow/late ESP-USB enumeration (BOT is slower than UAS)
# blows past systemd's 90s device-timeout into emergency mode with
# root locked — a dead end. nofail + short timeout lets the
# already-mounted root carry the boot; /boot mounts if/when it shows.
fileSystems."/boot".options = [ "nofail" "x-systemd.device-timeout=10s" ];
# NO growPartition/autoResize: sfdisk rewriting the partition table
# on first boot is the single most bus-stressing write, and flaky
# USB bridges drop off the bus mid-rewrite (sfdisk wedges in D-state
# and ESP-USB vanishes with the device). Persistent state is a few
# MB and the image ships ~2GB free. The internal-SATA disk-image-
# batm3 keeps growPartition (a real AHCI SSD won't drop the bus).
system.autoUpgrade.enable = lib.mkForce false;
})
];
};
};
# ── Standalone NixOS module ───────────────────────────────────
@ -443,41 +474,12 @@
# internal drive's ESP).
disk-image-batm3-usb =
let
cfg = self.nixosConfigurations.batm3-installed.extendModules {
modules = [
({ lib, ... }: {
fileSystems."/".device = lib.mkForce "/dev/disk/by-label/nixos-usb";
fileSystems."/boot".device = lib.mkForce "/dev/disk/by-label/ESP-USB";
# /boot must NOT be a hard boot dependency on the USB test
# image. The firmware already loaded the bootloader from the
# ESP before Linux started; /boot is only remounted so the OS
# can *update* the bootloader — which this image never does
# (autoUpgrade off, no nixos-rebuild on the stick). Without
# nofail, a slow/late ESP-USB enumeration (BOT is slower than
# UAS) blows past systemd's 90s device-timeout and drops to
# emergency mode — with root locked, an unrecoverable dead end.
# nofail + a short timeout lets the (already-mounted) root carry
# the boot to completion; /boot mounts if/when the ESP shows up.
fileSystems."/boot".options = [ "nofail" "x-systemd.device-timeout=10s" ];
# DELIBERATELY NO growPartition/autoResize on the USB image.
# growPartition runs sfdisk to rewrite the stick's partition
# table on first boot — the single most bus-stressing write of
# the boot. Flaky USB bridges drop off the bus mid-rewrite
# (sfdisk hangs forever as an uninterruptible D-state task) and,
# worse, partition 1 (ESP-USB) vanishes with the device, so
# /boot times out too. The kiosk's persistent state (state.db,
# .env, wifi.conf, logs) is a few MB and the built image already
# carries ~2GB free inside root — growing to fill the stick buys
# nothing and costs reliability. The internal-SATA target
# (disk-image-batm3) keeps growPartition: a real AHCI SSD won't
# drop the bus and there filling the disk is worth it.
system.autoUpgrade.enable = lib.mkForce false;
})
];
};
# Filesystem image of the batm3-usb config (defined in
# nixosConfigurations). Same config that in-place deploys target, so
# a reflash and a `switch-to-configuration` converge on one system.
baseImage = import (nixpkgs + "/nixos/lib/make-disk-image.nix") {
inherit pkgs lib;
config = cfg.config;
config = self.nixosConfigurations.batm3-usb.config;
format = "raw";
partitionTableType = "efi";
diskSize = "auto";