feat(machine): connectivity auto-recovery + on-screen Retry #82
6 changed files with 169 additions and 34 deletions
|
|
@ -404,6 +404,17 @@ ipcMain.handle('app:relaunch', (): void => {
|
|||
app.exit(0)
|
||||
})
|
||||
|
||||
// Connectivity recovery: reload the renderer to re-run init from a clean slate
|
||||
// (fresh JS context → no leaked actors/subscriptions), while preserving HAL in
|
||||
// this main process (reloadRenderer resets secretsConsumed so get-atm-secrets
|
||||
// works again, and hal:init is idempotent). The renderer calls this when it's
|
||||
// stuck on a connectivity-type "ATM Unavailable" and the network returns, or
|
||||
// when the operator taps the on-screen Retry (ADR-002 amendment 2026-08-04).
|
||||
ipcMain.handle('app:recover', (): void => {
|
||||
console.log('[Recovery] Reloading renderer to re-attempt initialization')
|
||||
reloadRenderer()
|
||||
})
|
||||
|
||||
// State persistence IPC handlers
|
||||
ipcMain.handle('state:load-cassettes', () => loadCassettes())
|
||||
ipcMain.handle('state:set-cassettes', (_event, cassettes) => setCassettes(cassettes))
|
||||
|
|
@ -480,6 +491,17 @@ let pendingBillDenomination: number | null = null
|
|||
|
||||
ipcMain.handle('hal:init', async (_event, config) => {
|
||||
try {
|
||||
// Idempotent: HAL lives in this (long-lived) main process, but the renderer
|
||||
// re-runs full init on every reload — the watchdog's crash-recovery reload
|
||||
// and the connectivity-recovery reload (app:recover) both re-invoke this.
|
||||
// initializeHal opens serial ports without closing prior handles, so
|
||||
// re-entering it would double-open the validator/dispenser. Reuse the
|
||||
// existing instance instead; its validator event wiring already targets the
|
||||
// (reloaded) mainWindow, so the reloaded renderer keeps receiving bill events.
|
||||
if (halInstance) {
|
||||
console.log('[Electron] HAL already initialized — reusing existing instance')
|
||||
return { success: true }
|
||||
}
|
||||
// Override cassette config with DB values (operator may have changed them via atm-tui
|
||||
// or via an operator-config publish from satmachineadmin). Pass per-position so the
|
||||
// HAL knows about every bay including duplicates of the same denomination — real
|
||||
|
|
|
|||
|
|
@ -124,6 +124,8 @@ contextBridge.exposeInMainWorld('electronAPI', {
|
|||
// then relaunch so the normal boot flow pairs it.
|
||||
saveSpireSeed: (seed: string): Promise<void> => ipcRenderer.invoke('state:save-spire-seed', seed),
|
||||
relaunchApp: (): Promise<void> => ipcRenderer.invoke('app:relaunch'),
|
||||
// Reload the renderer to re-attempt initialization (connectivity recovery).
|
||||
recoverApp: (): Promise<void> => ipcRenderer.invoke('app:recover'),
|
||||
|
||||
applyOperatorCassettesConfig: (
|
||||
payload: {
|
||||
|
|
@ -243,6 +245,7 @@ declare global {
|
|||
resetForRepair: () => Promise<void>
|
||||
saveSpireSeed: (seed: string) => Promise<void>
|
||||
relaunchApp: () => Promise<void>
|
||||
recoverApp: () => Promise<void>
|
||||
applyOperatorCassettesConfig: (
|
||||
payload: { positions: Record<string, { denomination: number; count: number }> },
|
||||
eventCreatedAt: number
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
<script setup lang="ts">
|
||||
import { onMounted, onUnmounted, ref, computed } from 'vue'
|
||||
import { onMounted, onUnmounted, ref, computed, watch } from 'vue'
|
||||
import { useRoute } from 'vue-router'
|
||||
import { useAtmStore } from '@/stores/atm'
|
||||
import { useTheme } from '@/composables/useTheme'
|
||||
|
|
@ -158,8 +158,71 @@ onMounted(async () => {
|
|||
|
||||
onUnmounted(() => {
|
||||
atmStore.stopPricePolling()
|
||||
stopRecoveryWatch()
|
||||
})
|
||||
|
||||
// ── Connectivity recovery (ADR-002 amendment 2026-08-04) ──────────────────
|
||||
// A connectivity-type init failure lands on "ATM Unavailable" and, without
|
||||
// this, stays there forever (init is one-shot; the nostr reconnect only helps
|
||||
// AFTER a first successful connect). We recover by reloading the renderer —
|
||||
// which re-runs this whole init from a clean JS context while the main process
|
||||
// keeps HAL (see main.ts app:recover). Not for the operator/self-clearing
|
||||
// states: `unpaired` shows the pairing wizard, `awaiting-fees` clears itself on
|
||||
// the operator's fee-config event, `maintenance` is operator-set.
|
||||
const NON_RECOVERABLE = new Set(['maintenance', 'awaiting-fees', 'unpaired'])
|
||||
const isRecoverable = computed(
|
||||
() => !!atmStore.initError && !NON_RECOVERABLE.has(atmStore.initError)
|
||||
)
|
||||
const recovering = ref(false)
|
||||
const RECOVERY_RETRY_MS = 45_000
|
||||
let recoveryTimer: ReturnType<typeof setInterval> | null = null
|
||||
|
||||
function triggerRecovery() {
|
||||
if (recovering.value) return
|
||||
recovering.value = true
|
||||
console.log('[App] Attempting connectivity recovery (renderer reload)')
|
||||
if (window.electronAPI?.recoverApp) {
|
||||
void window.electronAPI.recoverApp() // main reloads renderer → fresh init
|
||||
} else {
|
||||
location.reload() // browser-dev fallback
|
||||
}
|
||||
}
|
||||
|
||||
function onOnline() {
|
||||
// Network came back — recover immediately rather than waiting for the timer.
|
||||
triggerRecovery()
|
||||
}
|
||||
|
||||
function startRecoveryWatch() {
|
||||
stopRecoveryWatch()
|
||||
window.addEventListener('online', onOnline)
|
||||
// Safety net for the online-but-relay-unreachable case (navigator.onLine only
|
||||
// reflects a local route, not relay reachability).
|
||||
recoveryTimer = setInterval(triggerRecovery, RECOVERY_RETRY_MS)
|
||||
}
|
||||
|
||||
function stopRecoveryWatch() {
|
||||
window.removeEventListener('online', onOnline)
|
||||
if (recoveryTimer !== null) {
|
||||
clearInterval(recoveryTimer)
|
||||
recoveryTimer = null
|
||||
}
|
||||
}
|
||||
|
||||
/** Operator-facing "Retry" button on the maintenance screen. */
|
||||
function retryNow() {
|
||||
triggerRecovery()
|
||||
}
|
||||
|
||||
watch(
|
||||
isRecoverable,
|
||||
(recoverable) => {
|
||||
if (recoverable) startRecoveryWatch()
|
||||
else stopRecoveryWatch()
|
||||
},
|
||||
{ immediate: true }
|
||||
)
|
||||
|
||||
function toggleLiveServices() {
|
||||
if (atmStore.useLiveServices) {
|
||||
// Switch to mock
|
||||
|
|
@ -211,6 +274,19 @@ function toggleLiveServices() {
|
|||
>
|
||||
{{ atmStore.initError }}
|
||||
</p>
|
||||
|
||||
<!-- Manual recovery for a connectivity failure; auto-recovery also runs
|
||||
in the background (online event + backoff). Not shown for operator/
|
||||
self-clearing states (maintenance / awaiting-fees / unpaired). -->
|
||||
<Button
|
||||
v-if="isRecoverable"
|
||||
size="kiosk"
|
||||
:disabled="recovering"
|
||||
class="mt-4"
|
||||
@click="retryNow"
|
||||
>
|
||||
{{ recovering ? 'Retrying…' : 'Retry' }}
|
||||
</Button>
|
||||
</div>
|
||||
|
||||
<template v-else>
|
||||
|
|
|
|||
2
apps/machine/src/types/electron.d.ts
vendored
2
apps/machine/src/types/electron.d.ts
vendored
|
|
@ -101,6 +101,8 @@ declare global {
|
|||
resetForRepair: () => Promise<void>
|
||||
saveSpireSeed: (seed: string) => Promise<void>
|
||||
relaunchApp: () => Promise<void>
|
||||
/** Reload the renderer to re-attempt initialization (connectivity recovery). */
|
||||
recoverApp: () => Promise<void>
|
||||
applyOperatorCassettesConfig: (
|
||||
payload: { positions: Record<string, { denomination: number; count: number }> },
|
||||
eventCreatedAt: number
|
||||
|
|
|
|||
|
|
@ -104,6 +104,36 @@ The only honest way for a machine operator to exclude the SaaS operator is to **
|
|||
- The machine operator's Nostr key can become the single root of trust across all three planes — SSH `authorized_keys` + VPN enrollment at install, `AddOperator`/`RevokeOperator` (#42) for delegation — so granting/revoking any party (including the SaaS operator) is one scoped, revocable capability model.
|
||||
- `sshd` posture should be tightened to key-only for deployed boxes (password auth is currently forced on for installed configs for first-boot provisioning; scope it to the LAN/first-boot window). Tracks with [#51](https://git.atitlan.io/aiolabs/bitspire/issues/51).
|
||||
|
||||
## Amendment (2026-08-04): the access/recovery plane is not the *only* recovery
|
||||
|
||||
**Status:** Accepted · **Context:** the ATM app had no way to recover its own
|
||||
connectivity — a machine that booted with no internet (or whose init otherwise
|
||||
failed) sat on "ATM Unavailable" until a manual `systemctl restart bitspire`,
|
||||
even after the network came back.
|
||||
|
||||
This ADR's SSH/NetBird recovery plane stands — it is the operator's
|
||||
**app-and-OS-independent** path for the unanticipated and the broken, and may
|
||||
carry recovery *procedures* (restart the service, inspect logs, re-provision).
|
||||
But it is explicitly **not the first-line and not the only recovery method.**
|
||||
Recovery is layered, cheapest-first:
|
||||
|
||||
1. **App auto-recovery (first-line, no human).** The ATM app recovers its own
|
||||
relay/Lightning connectivity when possible: the nostr client already
|
||||
reconnects with backoff, and the app now re-initializes when connectivity
|
||||
returns (a fresh renderer reload — HAL is preserved in the main process),
|
||||
so "internet came back" self-heals without anyone touching the machine.
|
||||
2. **On-screen manual retry (operator at the machine).** The maintenance
|
||||
("ATM Unavailable") screen carries a **Retry** button so a person standing
|
||||
at the kiosk can force an immediate recovery attempt without shell access.
|
||||
3. **SSH/NetBird (operator remote, last resort).** This plane — for when the
|
||||
app *can't* self-heal or the box is genuinely broken. Unchanged by this
|
||||
amendment beyond the reframing: it is the floor, not the front line.
|
||||
|
||||
Rationale: the common failure (transient network / boot-before-network) must
|
||||
not require remote shell access to a public kiosk. Reserve the heavyweight
|
||||
recovery plane for genuine app/OS failure. Implemented on branch
|
||||
`feat/connection-recovery`.
|
||||
|
||||
## References
|
||||
|
||||
- [#41](https://git.atitlan.io/aiolabs/bitspire/issues/41) — Multi-location deployment: runtime site config (the access plane's per-machine identity is provisioned here, not baked into the closure).
|
||||
|
|
|
|||
68
flake.nix
68
flake.nix
|
|
@ -292,6 +292,37 @@
|
|||
# at ttyJ5, dispenser at ttyJ7 layout) — reuse the same hw module.
|
||||
sintra-installed = mkInstalledConfig "sintra" ./deploy/nixos/hardware/upboard.nix;
|
||||
batm3-installed = mkInstalledConfig "batm3" ./deploy/nixos/hardware/batm3.nix;
|
||||
|
||||
# USB-bootable variant of batm3-installed. This is the config the
|
||||
# flashed USB stick actually runs — distinct fs labels so stage-1 can't
|
||||
# latch the internal drive, nofail /boot, no growPartition, autoUpgrade
|
||||
# off. Exposed as a named config (not just inline in the disk-image
|
||||
# target) so its system closure can be built here and deployed in-place
|
||||
# with `nix copy` + `switch-to-configuration` — updating the app on a
|
||||
# running stick WITHOUT reflashing (preserves pairing + /var/lib state).
|
||||
# disk-image-batm3-usb builds its filesystem image from this same config.
|
||||
batm3-usb = self.nixosConfigurations.batm3-installed.extendModules {
|
||||
modules = [
|
||||
({ lib, ... }: {
|
||||
fileSystems."/".device = lib.mkForce "/dev/disk/by-label/nixos-usb";
|
||||
fileSystems."/boot".device = lib.mkForce "/dev/disk/by-label/ESP-USB";
|
||||
# /boot must NOT be a hard boot dependency on the USB image. The
|
||||
# firmware already loaded the bootloader before Linux; without
|
||||
# nofail, a slow/late ESP-USB enumeration (BOT is slower than UAS)
|
||||
# blows past systemd's 90s device-timeout into emergency mode with
|
||||
# root locked — a dead end. nofail + short timeout lets the
|
||||
# already-mounted root carry the boot; /boot mounts if/when it shows.
|
||||
fileSystems."/boot".options = [ "nofail" "x-systemd.device-timeout=10s" ];
|
||||
# NO growPartition/autoResize: sfdisk rewriting the partition table
|
||||
# on first boot is the single most bus-stressing write, and flaky
|
||||
# USB bridges drop off the bus mid-rewrite (sfdisk wedges in D-state
|
||||
# and ESP-USB vanishes with the device). Persistent state is a few
|
||||
# MB and the image ships ~2GB free. The internal-SATA disk-image-
|
||||
# batm3 keeps growPartition (a real AHCI SSD won't drop the bus).
|
||||
system.autoUpgrade.enable = lib.mkForce false;
|
||||
})
|
||||
];
|
||||
};
|
||||
};
|
||||
|
||||
# ── Standalone NixOS module ───────────────────────────────────
|
||||
|
|
@ -443,41 +474,12 @@
|
|||
# internal drive's ESP).
|
||||
disk-image-batm3-usb =
|
||||
let
|
||||
cfg = self.nixosConfigurations.batm3-installed.extendModules {
|
||||
modules = [
|
||||
({ lib, ... }: {
|
||||
fileSystems."/".device = lib.mkForce "/dev/disk/by-label/nixos-usb";
|
||||
fileSystems."/boot".device = lib.mkForce "/dev/disk/by-label/ESP-USB";
|
||||
# /boot must NOT be a hard boot dependency on the USB test
|
||||
# image. The firmware already loaded the bootloader from the
|
||||
# ESP before Linux started; /boot is only remounted so the OS
|
||||
# can *update* the bootloader — which this image never does
|
||||
# (autoUpgrade off, no nixos-rebuild on the stick). Without
|
||||
# nofail, a slow/late ESP-USB enumeration (BOT is slower than
|
||||
# UAS) blows past systemd's 90s device-timeout and drops to
|
||||
# emergency mode — with root locked, an unrecoverable dead end.
|
||||
# nofail + a short timeout lets the (already-mounted) root carry
|
||||
# the boot to completion; /boot mounts if/when the ESP shows up.
|
||||
fileSystems."/boot".options = [ "nofail" "x-systemd.device-timeout=10s" ];
|
||||
# DELIBERATELY NO growPartition/autoResize on the USB image.
|
||||
# growPartition runs sfdisk to rewrite the stick's partition
|
||||
# table on first boot — the single most bus-stressing write of
|
||||
# the boot. Flaky USB bridges drop off the bus mid-rewrite
|
||||
# (sfdisk hangs forever as an uninterruptible D-state task) and,
|
||||
# worse, partition 1 (ESP-USB) vanishes with the device, so
|
||||
# /boot times out too. The kiosk's persistent state (state.db,
|
||||
# .env, wifi.conf, logs) is a few MB and the built image already
|
||||
# carries ~2GB free inside root — growing to fill the stick buys
|
||||
# nothing and costs reliability. The internal-SATA target
|
||||
# (disk-image-batm3) keeps growPartition: a real AHCI SSD won't
|
||||
# drop the bus and there filling the disk is worth it.
|
||||
system.autoUpgrade.enable = lib.mkForce false;
|
||||
})
|
||||
];
|
||||
};
|
||||
# Filesystem image of the batm3-usb config (defined in
|
||||
# nixosConfigurations). Same config that in-place deploys target, so
|
||||
# a reflash and a `switch-to-configuration` converge on one system.
|
||||
baseImage = import (nixpkgs + "/nixos/lib/make-disk-image.nix") {
|
||||
inherit pkgs lib;
|
||||
config = cfg.config;
|
||||
config = self.nixosConfigurations.batm3-usb.config;
|
||||
format = "raw";
|
||||
partitionTableType = "efi";
|
||||
diskSize = "auto";
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue