Compare commits
2 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| ea8fe22969 | |||
| 2910672b5a |
@@ -0,0 +1,139 @@
|
||||
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import { eq, siteConfig } from "@parking/db";
|
||||
import { createTestDb } from "@parking/db/testing";
|
||||
import { afterEach, beforeEach, describe, expect, it } from "vitest";
|
||||
import { BackupService } from "./backup-service.js";
|
||||
|
||||
// BackupService previously tracked last-success/last-error as plain in-process fields, so a
|
||||
// server restart (a fresh BackupService instance, exactly as happens on every deploy/crash/OOM
|
||||
// reboot under `restart: always`) silently reset the admin UI to "last successful backup:
|
||||
// Never" — even with valid, correctly-rotating backups already on disk (2026-08-30 field
|
||||
// incident, park-buzi). These tests exercise the fix: status is read from site_config, so a new
|
||||
// BackupService instance pointed at the same DB sees the prior instance's last-run outcome, and
|
||||
// the schedule is wall-clock-based (isDue()) rather than time-since-process-start.
|
||||
// See wiki/concepts/backup-recovery.md.
|
||||
|
||||
const KEY = "a-test-backup-key-that-is-long-enough";
|
||||
|
||||
let workDir: string;
|
||||
let target: string;
|
||||
|
||||
beforeEach(() => {
|
||||
workDir = mkdtempSync(join(tmpdir(), "pk-backup-service-test-"));
|
||||
target = join(workDir, "target");
|
||||
process.env.BACKUP_KEY = KEY;
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
rmSync(workDir, { recursive: true, force: true });
|
||||
delete process.env.BACKUP_KEY;
|
||||
});
|
||||
|
||||
function setTargetDir(db: ReturnType<typeof createTestDb>["db"], dir: string): void {
|
||||
const existing = db.select().from(siteConfig).where(eq(siteConfig.id, 1)).get();
|
||||
if (existing) {
|
||||
db.update(siteConfig).set({ backupTargetDir: dir }).where(eq(siteConfig.id, 1)).run();
|
||||
} else {
|
||||
db.insert(siteConfig).values({ id: 1, backupTargetDir: dir }).run();
|
||||
}
|
||||
}
|
||||
|
||||
describe("BackupService — persisted status survives a restart", () => {
|
||||
it("a fresh instance sees the previous instance's last success", async () => {
|
||||
const t = createTestDb();
|
||||
setTargetDir(t.db, target);
|
||||
|
||||
const first = new BackupService(t.db);
|
||||
expect(first.status().lastSuccessAt).toBeNull();
|
||||
const result = await first.run("manual");
|
||||
|
||||
// Simulate a process restart: a brand-new BackupService over the SAME db handle (in
|
||||
// production this would be a fresh process re-opening the same sqlite file).
|
||||
const second = new BackupService(t.db);
|
||||
const status = second.status();
|
||||
expect(status.lastSuccessAt).not.toBeNull();
|
||||
expect(status.lastResult).toEqual({ path: result.path, bytes: result.bytes, prunedFiles: result.prunedFiles });
|
||||
expect(status.lastError).toBeNull();
|
||||
|
||||
t.close();
|
||||
});
|
||||
|
||||
it("a fresh instance sees the previous instance's last error, and it clears on next success", async () => {
|
||||
const t = createTestDb();
|
||||
// Target dir set, but as a FILE (not a directory) — runBackup's mkdir(recursive) will
|
||||
// throw, giving us a real, deterministic failure without needing to mock anything.
|
||||
const badTarget = join(workDir, "not-a-dir");
|
||||
writeFileSync(badTarget, "x");
|
||||
setTargetDir(t.db, badTarget);
|
||||
|
||||
const first = new BackupService(t.db);
|
||||
await expect(first.run("manual")).rejects.toThrow();
|
||||
|
||||
const second = new BackupService(t.db);
|
||||
const status = second.status();
|
||||
expect(status.lastError).not.toBeNull();
|
||||
expect(status.lastErrorAt).not.toBeNull();
|
||||
expect(status.lastSuccessAt).toBeNull();
|
||||
|
||||
// Now point at a real directory and succeed — the persisted error must clear.
|
||||
setTargetDir(t.db, target);
|
||||
await second.run("manual");
|
||||
const third = new BackupService(t.db);
|
||||
const finalStatus = third.status();
|
||||
expect(finalStatus.lastSuccessAt).not.toBeNull();
|
||||
expect(finalStatus.lastError).toBeNull();
|
||||
expect(finalStatus.lastErrorAt).toBeNull();
|
||||
|
||||
t.close();
|
||||
});
|
||||
});
|
||||
|
||||
describe("BackupService — isDue() is wall-clock-based, not process-uptime-based", () => {
|
||||
it("is due immediately when no success has ever been recorded", () => {
|
||||
const t = createTestDb();
|
||||
const svc = new BackupService(t.db);
|
||||
expect(svc.isDue()).toBe(true);
|
||||
t.close();
|
||||
});
|
||||
|
||||
it("is NOT due right after a fresh instance is constructed, if a recent success is persisted", async () => {
|
||||
const t = createTestDb();
|
||||
setTargetDir(t.db, target);
|
||||
const first = new BackupService(t.db);
|
||||
await first.run("manual");
|
||||
|
||||
// The whole point of the fix: a brand-new instance (simulating a restart moments after a
|
||||
// real backup completed) must NOT think a backup is due just because ITS OWN uptime is ~0.
|
||||
const second = new BackupService(t.db);
|
||||
expect(second.isDue()).toBe(false);
|
||||
t.close();
|
||||
});
|
||||
|
||||
it("is due once the persisted last-success timestamp is old enough", async () => {
|
||||
const t = createTestDb();
|
||||
setTargetDir(t.db, target);
|
||||
const svc = new BackupService(t.db);
|
||||
await svc.run("manual");
|
||||
|
||||
const almostADayLater = new Date(Date.now() + 23 * 60 * 60 * 1000);
|
||||
expect(svc.isDue(almostADayLater)).toBe(false);
|
||||
|
||||
const overADayLater = new Date(Date.now() + 24 * 60 * 60 * 1000 + 1000);
|
||||
expect(svc.isDue(overADayLater)).toBe(true);
|
||||
t.close();
|
||||
});
|
||||
|
||||
it("runScheduled() is a no-op when not yet due, even if configured", async () => {
|
||||
const t = createTestDb();
|
||||
setTargetDir(t.db, target);
|
||||
const svc = new BackupService(t.db);
|
||||
await svc.run("manual");
|
||||
const afterFirst = svc.status().lastSuccessAt;
|
||||
|
||||
await svc.runScheduled(); // not due yet — must not run again
|
||||
expect(svc.status().lastSuccessAt).toBe(afterFirst);
|
||||
t.close();
|
||||
});
|
||||
});
|
||||
@@ -11,6 +11,12 @@ import { DEFAULT_BACKUP_RETENTION, runBackup, type BackupResult, type BackupRete
|
||||
// a key must never live in the DB it backs up. Remembers the last outcome so the route + UI can
|
||||
// show last-success / last-error, and serializes concurrent runs (manual + timer). See
|
||||
// wiki/concepts/backup-recovery.md.
|
||||
//
|
||||
// Last-success/last-error are PERSISTED to site_config (backup_last_*), not just held in
|
||||
// memory — an earlier version tracked these as plain in-process fields only, so every server
|
||||
// restart (deploy, crash, OOM, host reboot — all routine under `restart: always`) silently
|
||||
// reset the admin UI to "last successful backup: Never", even with valid, correctly-rotating
|
||||
// backups already on disk (2026-08-30 field incident, park-buzi). See wiki/concepts/backup-recovery.md.
|
||||
|
||||
/** The dedicated backup-encryption key, from env (NOT the DB). Separate from EVENT_SIGNING_KEY. */
|
||||
export function backupKeyFromEnv(): string {
|
||||
@@ -65,16 +71,33 @@ export class BackupService {
|
||||
readonly #logger?: FastifyBaseLogger;
|
||||
|
||||
#running = false;
|
||||
#lastSuccessAt: string | null = null;
|
||||
#lastResult: BackupResult | null = null;
|
||||
#lastErrorAt: string | null = null;
|
||||
#lastError: string | null = null;
|
||||
|
||||
constructor(db: Db, logger?: FastifyBaseLogger) {
|
||||
this.#db = db;
|
||||
this.#logger = logger;
|
||||
}
|
||||
|
||||
/** Fresh read of the persisted row (single source of truth — no in-memory cache to go stale
|
||||
* or reset on restart). */
|
||||
#row(): { backupLastSuccessAt: string | null; backupLastResultJson: string | null; backupLastErrorAt: string | null; backupLastError: string | null } | undefined {
|
||||
return this.#db.select().from(siteConfig).where(eq(siteConfig.id, 1)).get();
|
||||
}
|
||||
|
||||
#persist(patch: {
|
||||
backupLastSuccessAt?: string | null;
|
||||
backupLastResultJson?: string | null;
|
||||
backupLastErrorAt?: string | null;
|
||||
backupLastError?: string | null;
|
||||
}): void {
|
||||
const updatedAt = new Date().toISOString();
|
||||
const existing = this.#row();
|
||||
if (existing) {
|
||||
this.#db.update(siteConfig).set({ ...patch, updatedAt }).where(eq(siteConfig.id, 1)).run();
|
||||
} else {
|
||||
this.#db.insert(siteConfig).values({ id: 1, ...patch, updatedAt }).run();
|
||||
}
|
||||
}
|
||||
|
||||
/** The admin-chosen target dir from site_config (null/empty = unset). Read fresh each call. */
|
||||
targetDir(): string | null {
|
||||
const row = this.#db.select().from(siteConfig).where(eq(siteConfig.id, 1)).get();
|
||||
@@ -104,6 +127,15 @@ export class BackupService {
|
||||
|
||||
status(): BackupStatus {
|
||||
const r = this.retention();
|
||||
const row = this.#row();
|
||||
let lastResult: BackupStatus["lastResult"] = null;
|
||||
if (row?.backupLastResultJson) {
|
||||
try {
|
||||
lastResult = JSON.parse(row.backupLastResultJson) as BackupStatus["lastResult"];
|
||||
} catch {
|
||||
lastResult = null; // corrupt/foreign value in the column — don't let it crash status()
|
||||
}
|
||||
}
|
||||
return {
|
||||
configured: this.configured,
|
||||
targetDir: this.targetDir(),
|
||||
@@ -111,12 +143,10 @@ export class BackupService {
|
||||
keepDailyDays: r.keepDailyDays,
|
||||
keyPresent: this.keyPresent,
|
||||
running: this.#running,
|
||||
lastSuccessAt: this.#lastSuccessAt,
|
||||
lastResult: this.#lastResult
|
||||
? { path: this.#lastResult.path, bytes: this.#lastResult.bytes, prunedFiles: this.#lastResult.prunedFiles }
|
||||
: null,
|
||||
lastErrorAt: this.#lastErrorAt,
|
||||
lastError: this.#lastError,
|
||||
lastSuccessAt: row?.backupLastSuccessAt ?? null,
|
||||
lastResult,
|
||||
lastErrorAt: row?.backupLastErrorAt ?? null,
|
||||
lastError: row?.backupLastError ?? null,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -139,14 +169,17 @@ export class BackupService {
|
||||
try {
|
||||
this.#logger?.info(`backup: starting (${trigger}) → ${targetDir}`);
|
||||
const res = await runBackup(this.#db, { targetDir, key, retention: this.retention() }, this.#logger);
|
||||
this.#lastResult = res;
|
||||
this.#lastSuccessAt = new Date().toISOString();
|
||||
this.#lastError = null;
|
||||
this.#persist({
|
||||
backupLastSuccessAt: new Date().toISOString(),
|
||||
backupLastResultJson: JSON.stringify({ path: res.path, bytes: res.bytes, prunedFiles: res.prunedFiles }),
|
||||
backupLastErrorAt: null,
|
||||
backupLastError: null,
|
||||
});
|
||||
return res;
|
||||
} catch (err) {
|
||||
this.#lastError = (err as Error).message;
|
||||
this.#lastErrorAt = new Date().toISOString();
|
||||
this.#logger?.error(`backup: failed (${trigger}): ${this.#lastError}`);
|
||||
const message = (err as Error).message;
|
||||
this.#persist({ backupLastErrorAt: new Date().toISOString(), backupLastError: message });
|
||||
this.#logger?.error(`backup: failed (${trigger}): ${message}`);
|
||||
throw err;
|
||||
} finally {
|
||||
this.#running = false;
|
||||
@@ -156,13 +189,34 @@ export class BackupService {
|
||||
return this.#inflight;
|
||||
}
|
||||
|
||||
/** Scheduled-run wrapper: never throws (a timer must not crash the process). */
|
||||
/**
|
||||
* Scheduled-run wrapper: never throws (a timer must not crash the process). Safe to call on
|
||||
* a short, frequent poll (see server.ts) — it's a no-op unless `isDue()` says a full interval
|
||||
* has actually elapsed since the last recorded success, so frequent polling doesn't cause
|
||||
* frequent backups.
|
||||
*/
|
||||
async runScheduled(): Promise<void> {
|
||||
if (!this.configured) return; // silent no-op when backups aren't set up
|
||||
if (!this.isDue()) return;
|
||||
try {
|
||||
await this.run("scheduled");
|
||||
} catch {
|
||||
/* recorded in last-error; already logged */
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Wall-clock check: has enough time elapsed since the last successful backup for a new one
|
||||
* to be due? Deliberately based on the PERSISTED last-success instant, not "time since this
|
||||
* process started" — a `setInterval(..., 24h)` measured from process start silently drifts
|
||||
* (or skips a whole day) across every restart, since the countdown restarts from zero each
|
||||
* time regardless of when the last real backup happened. See wiki/concepts/backup-recovery.md.
|
||||
*/
|
||||
isDue(now: Date = new Date(), intervalMs = 24 * 60 * 60 * 1000): boolean {
|
||||
const lastSuccessAt = this.#row()?.backupLastSuccessAt;
|
||||
if (!lastSuccessAt) return true; // never recorded a success → due immediately once configured
|
||||
const last = new Date(lastSuccessAt).getTime();
|
||||
if (Number.isNaN(last)) return true;
|
||||
return now.getTime() - last >= intervalMs;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -336,12 +336,20 @@ export async function buildServer(opts: BuildOptions = {}): Promise<FastifyInsta
|
||||
void runSnapPrune(); // once at startup
|
||||
app.addHook("onClose", async () => clearInterval(snapPruneTimer));
|
||||
|
||||
// Scheduled encrypted backup — daily, unref'd. A no-op (silent) until BACKUP_TARGET_DIR +
|
||||
// BACKUP_KEY are configured; tolerates an unreachable/unmounted target by recording the
|
||||
// error and trying again next run. NOT run once at startup (a just-booted appliance after a
|
||||
// power cut shouldn't immediately write to a possibly-not-yet-mounted disk; the daily cadence
|
||||
// and the manual button cover it). See wiki/concepts/backup-recovery.md.
|
||||
const backupTimer = setInterval(() => void backupService.runScheduled(), 24 * 60 * 60 * 1000);
|
||||
// Scheduled encrypted backup — checked every 15 min, unref'd; `runScheduled()` itself is a
|
||||
// no-op unless a full 24h has actually elapsed since the last PERSISTED success (isDue(), in
|
||||
// backup-service.ts), so this frequent poll does not cause frequent backups. Deliberately
|
||||
// NOT a `setInterval(..., 24h)` measured from process start: that design silently reset its
|
||||
// own countdown on every restart (deploy/crash/OOM/reboot, all routine under `restart:
|
||||
// always`), which could push a day's backup out arbitrarily far AND — before last-success was
|
||||
// persisted — made the admin UI show "Never" despite valid backups already on disk
|
||||
// (2026-08-30 field incident, park-buzi). A short poll against a persisted, wall-clock
|
||||
// timestamp is immune to both restart timing and to any single restart cadence. A no-op
|
||||
// (silent) until BACKUP_TARGET_DIR + BACKUP_KEY are configured; tolerates an
|
||||
// unreachable/unmounted target by recording the error and trying again next check. NOT run
|
||||
// once at startup (a just-booted appliance after a power cut shouldn't immediately write to a
|
||||
// possibly-not-yet-mounted disk). See wiki/concepts/backup-recovery.md.
|
||||
const backupTimer = setInterval(() => void backupService.runScheduled(), 15 * 60 * 1000);
|
||||
backupTimer.unref();
|
||||
app.addHook("onClose", async () => clearInterval(backupTimer));
|
||||
if (backupService.configured) {
|
||||
|
||||
@@ -64,6 +64,39 @@ EVENT_SIGNING_KEY=[[park_lab_event_signing_key]]
|
||||
BACKUP_KEY=[[park_lab_backup_key]]
|
||||
"""
|
||||
|
||||
##############################################################################
|
||||
# art-docker-station — second LAB bench box (hardware/dev testing, no real traffic). Same tier as
|
||||
# park-lab: chases `dev` (compose files + MOVING image tag), own art_docker_station_* secret refs
|
||||
# (never shared with park-lab or a real booth, even lab-to-lab — per-box blast radius).
|
||||
##############################################################################
|
||||
|
||||
[[stack]]
|
||||
name = "art-docker-station"
|
||||
[stack.config]
|
||||
server = "art-docker-station"
|
||||
git_provider = "git.infra.msai.al"
|
||||
git_account = "komodo"
|
||||
repo = "mca/parking_solution"
|
||||
branch = "dev"
|
||||
file_paths = [
|
||||
"docker-compose.yml",
|
||||
"docker-compose.prod.yml"
|
||||
]
|
||||
registry_provider = "git.infra.msai.al"
|
||||
registry_account = "komodo"
|
||||
environment = """
|
||||
REGISTRY=git.infra.msai.al/mca/parking_solution
|
||||
# Lab tier: the MOVING dev tag — redeploy pulls the latest dev build. Pin to a
|
||||
# dev-<sha> only when reproducing a specific state.
|
||||
TAG=dev
|
||||
COOKIE_SECURE=0
|
||||
VISION_ENABLED=1
|
||||
WS_ALLOWED_ORIGINS=
|
||||
JWT_SECRET=[[art_docker_station_jwt_secret]]
|
||||
EVENT_SIGNING_KEY=[[art_docker_station_event_signing_key]]
|
||||
BACKUP_KEY=[[art_docker_station_backup_key]]
|
||||
"""
|
||||
|
||||
##############################################################################
|
||||
|
||||
[[stack]]
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
-- Last-success/last-error for the encrypted DB backup were previously tracked only as
|
||||
-- in-process fields on BackupService (never written to the DB) — so every server restart
|
||||
-- (deploy/crash/OOM/host reboot, all routine under `restart: always`) silently reset the admin
|
||||
-- UI's "last successful backup" to "Never", even with valid, correctly-rotating backups already
|
||||
-- on disk (2026-08-30 field incident, park-buzi). Four additive, nullable columns; null = no
|
||||
-- run recorded yet (or, for the error pair, no failure since the last success). See
|
||||
-- wiki/concepts/backup-recovery.md.
|
||||
ALTER TABLE `site_config` ADD `backup_last_success_at` text;--> statement-breakpoint
|
||||
ALTER TABLE `site_config` ADD `backup_last_result_json` text;--> statement-breakpoint
|
||||
ALTER TABLE `site_config` ADD `backup_last_error_at` text;--> statement-breakpoint
|
||||
ALTER TABLE `site_config` ADD `backup_last_error` text;
|
||||
@@ -176,6 +176,13 @@
|
||||
"when": 1783948800000,
|
||||
"tag": "0024_validation_programs",
|
||||
"breakpoints": true
|
||||
},
|
||||
{
|
||||
"idx": 25,
|
||||
"version": "6",
|
||||
"when": 1788078414270,
|
||||
"tag": "0025_backup_last_status",
|
||||
"breakpoints": true
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -288,6 +288,20 @@ export const siteConfig = sqliteTable("site_config", {
|
||||
backupKeepLast: integer("backup_keep_last"),
|
||||
/** Beyond keepLast, keep one backup per day for this many days. null ⇒ code default (30). */
|
||||
backupKeepDailyDays: integer("backup_keep_daily_days"),
|
||||
/** ISO timestamp of the last backup that actually completed successfully. Persisted here
|
||||
* (not just in-process memory) so the admin UI's "last successful backup" survives a
|
||||
* server restart — before this column existed, a restart silently reset that status to
|
||||
* "Never" even with valid backups already on disk. null = no successful run recorded yet.
|
||||
* See wiki/concepts/backup-recovery.md. */
|
||||
backupLastSuccessAt: text("backup_last_success_at"),
|
||||
/** JSON-encoded { path, bytes, prunedFiles } of the last successful run, for the same
|
||||
* restart-durability reason as backupLastSuccessAt. null = none recorded yet. */
|
||||
backupLastResultJson: text("backup_last_result_json"),
|
||||
/** ISO timestamp of the last FAILED scheduled/manual backup attempt, persisted for the same
|
||||
* reason. null = no failure recorded (or none since the last success). */
|
||||
backupLastErrorAt: text("backup_last_error_at"),
|
||||
/** Error message of the last failed attempt. Cleared (set null) on the next success. */
|
||||
backupLastError: text("backup_last_error"),
|
||||
updatedAt: text("updated_at")
|
||||
.notNull()
|
||||
.default(sql`(current_timestamp)`),
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
type: concept
|
||||
tags: [parking, durability, backup, recovery, security, crypto]
|
||||
sources: []
|
||||
updated: 2026-06-29
|
||||
updated: 2026-08-30
|
||||
---
|
||||
|
||||
# Backup & Disaster Recovery
|
||||
@@ -207,12 +207,68 @@ timer + the manual route**. What landed:
|
||||
**SMB/NFS already work** — they're just a mounted path the admin enters as the target. **Deferred to
|
||||
follow-up slices:** an **SFTP** target and a **restore runbook / CLI**.
|
||||
|
||||
## Field bug — "last successful backup: Never" despite valid, rotating backups on disk (found + fixed 2026-08-30)
|
||||
|
||||
**Symptom (park-buzi):** the admin noticed the backup directory held 7 real, correctly-sized,
|
||||
correctly-rotating encrypted backups (`parking-backup-*.sqlite.enc`, retention working exactly as
|
||||
designed) — yet the Backup screen's "Kopja e fundit e suksesshme" (last successful backup) showed
|
||||
**"Asnjëherë" (Never)**. Separately, the most recent file was 2 days old rather than ~1.
|
||||
|
||||
**Root cause — two independent, disconnected code paths, both traced to `setInterval`-since-
|
||||
process-start:**
|
||||
|
||||
1. **Status was never persisted.** `BackupService` tracked `lastSuccessAt`/`lastResult`/
|
||||
`lastErrorAt`/`lastError` as **plain in-process private fields** — set only inside `run()`,
|
||||
read only by `status()` on the *same running instance*. Nothing wrote them to `site_config` or
|
||||
anywhere else durable. The actual backup-writing engine (`backup.ts`: consistent copy → encrypt
|
||||
→ `pruneOldBackups`) is a completely separate code path that only touches the filesystem and
|
||||
has no notion of this status object. So "7 valid files on disk" and "status says Never" were
|
||||
never contradictory — they were two unrelated signals, and **any** server restart (deploy,
|
||||
crash, OOM, host reboot — all routine under `restart: always` in `docker-compose.prod.yml`)
|
||||
silently reset the in-memory fields to `null` regardless of what had actually happened on disk.
|
||||
2. **The schedule was measured from process start, not from the last real backup.** The daily
|
||||
timer was `setInterval(() => backupService.runScheduled(), 24h)` — a fixed 24h period counted
|
||||
from whenever the *process* last started, not from wall-clock time or from when a backup last
|
||||
actually succeeded. The exact same restart that wiped the in-memory status also reset this
|
||||
countdown, which is why the cadence can silently drift or skip past a day with no error ever
|
||||
surfacing anywhere.
|
||||
|
||||
Both symptoms are one cause: **the server process restarted after the Aug 28 backup, and nothing
|
||||
about this design was built to survive that.**
|
||||
|
||||
### Fix (2026-08-30)
|
||||
|
||||
- **`packages/db/src/schema.ts`** / migration `0025_backup_last_status.sql` — four new nullable
|
||||
`site_config` columns: `backup_last_success_at`, `backup_last_result_json`,
|
||||
`backup_last_error_at`, `backup_last_error`. Same table, same upsert pattern as
|
||||
`backup_target_dir`/`backup_keep_last`/`backup_keep_daily_days` (migrations 0016/0017).
|
||||
- **`backup-service.ts`** — `run()` now writes success/error outcomes to these columns (via a
|
||||
`#persist` upsert helper) instead of private fields; `status()` reads them fresh from the DB on
|
||||
every call. A brand-new `BackupService` instance (i.e. a fresh process) now sees exactly what
|
||||
the previous instance last recorded — no more restart amnesia.
|
||||
- **New `isDue(now, intervalMs = 24h)`** method: due iff `now - backupLastSuccessAt >= 24h` (or
|
||||
immediately due if no success was ever recorded), computed from the **persisted** timestamp —
|
||||
never from process uptime.
|
||||
- **`server.ts`** — the daily `setInterval` was replaced with a **15-minute poll** calling
|
||||
`runScheduled()`, which now itself no-ops unless `isDue()` is true. This makes the actual backup
|
||||
cadence immune to restart timing entirely: however often the process happens to restart, the
|
||||
next backup fires within 15 minutes of 24h having genuinely elapsed since the last real success
|
||||
— not 24h after whatever moment the process most recently came back up.
|
||||
- Covered by a new `backup-service.test.ts`: a fresh `BackupService` over the same DB handle
|
||||
(simulating a restart) sees the prior instance's last success/error and its cleared-on-success
|
||||
behavior; `isDue()` is exercised directly against injected timestamps rather than real sleeps.
|
||||
|
||||
No change to the `BackupStatus` shape returned by `GET /api/backup/status` or to
|
||||
`BackupSettings.tsx` — this was purely a durability fix underneath the same contract.
|
||||
|
||||
## Status
|
||||
|
||||
Design settled 2026-06-29; **engine + admin-configured local/mounted target + admin UI BUILT
|
||||
2026-06-29** (SFTP + restore tooling pending). The target directory is **admin-chosen in the UI**
|
||||
(`site_config`, migration 0016), not an env var — the on-site admin picks where backups land; only
|
||||
`BACKUP_KEY` stays a server secret. Resolves the *design* half of [[open-questions]] #5 and the first
|
||||
build slices; records the key-custody stance that bears on #6 (signing stays decoupled from the TPM) and
|
||||
#10 (snapshots bloat backups → future exclude toggle). See [[append-only-event-chain]],
|
||||
[[disk-os-hardening]], [[tpm]], [[fleet-deployment-komodo]], [[reconciliation]].
|
||||
`BACKUP_KEY` stays a server secret. **Last-success/last-error status + the scheduling cadence are
|
||||
now restart-durable (migration 0025, 2026-08-30)** — see field bug above. Resolves the *design*
|
||||
half of [[open-questions]] #5 and the first build slices; records the key-custody stance that bears
|
||||
on #6 (signing stays decoupled from the TPM) and #10 (snapshots bloat backups → future exclude
|
||||
toggle). See [[append-only-event-chain]], [[disk-os-hardening]], [[tpm]], [[fleet-deployment-komodo]],
|
||||
[[reconciliation]].
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
type: concept
|
||||
tags: [parking, device, printer, transport, usb, escpos, provisioning]
|
||||
sources: []
|
||||
updated: 2026-07-06
|
||||
updated: 2026-08-30
|
||||
status: settled
|
||||
---
|
||||
|
||||
@@ -140,5 +140,70 @@ hint. The transport option label no longer hardcodes lp0.
|
||||
> the monitor would mark a perfectly working printer offline/degraded. Over USB the two drivers
|
||||
> behave identically (reachability floor), so either works post-fix. See [[rongta-printer]].
|
||||
|
||||
## Field bug — cover-open re-enumeration wedges the container's `/dev/usb` view; only `docker restart`, not a host reboot, clears it (investigated 2026-08-30, unconfirmed root cause)
|
||||
|
||||
**Symptom (park-buzi, unknown/"Generic" USB printer, model not yet identified — see below):** every
|
||||
time the booth operator opens the printer's paper-roll cover to reload paper, the printer's status
|
||||
goes `offline`/faulty in the app and **never self-recovers** — not after the cover closes, not after
|
||||
a full appliance reboot. The only fix found so far is SSH in and `docker restart server`.
|
||||
|
||||
**Ruled out at the application layer.** Traced `sendRawUsb`/`probeUsb` in `printer-escpos.ts`: every
|
||||
print AND every poll tick (`device-monitor.ts` 8s / `printer-monitor.ts` 5s) does a fresh
|
||||
`open()` → write/probe → `close()` against the configured `devicePath`. **No fd, socket, or driver
|
||||
instance is held across calls** — `driver.create(config)` is a throwaway object with no persistent
|
||||
handle. So a naive "stale Node file descriptor" explanation does not fit this codebase; the
|
||||
app-layer retry-by-fresh-open-every-poll should self-heal within one poll cycle if the kernel's view
|
||||
of the device node is current.
|
||||
|
||||
**Leading hypothesis: the container's bind-mount of `/dev/usb`, not the Node process, holds the
|
||||
stale state.** Docker Compose wires the printer in as a **directory bind-mount**
|
||||
(`docker-compose.prod.yml`, `volumes: - /dev/usb:/dev/usb`), chosen deliberately (per its own
|
||||
comment) so the app survives the printer renumbering to a different `lpN`. But many USB thermal
|
||||
printers cut power to their own USB interface board when the cover-open microswitch trips (a
|
||||
hardware safety/power feature, not just a status flag) — the printer drops off the bus and
|
||||
re-enumerates, potentially as a new device node, when the cover closes. The **host** kernel picks
|
||||
this up fine; the **container's mount namespace**, once established, is a known Docker/OverlayFS
|
||||
sharp edge for `/dev` subtree bind-mounts — it can keep resolving the old node until the mount
|
||||
itself is redone.
|
||||
|
||||
- `docker restart server` recreates the container's mount namespace → the `/dev/usb` bind-mount is
|
||||
redone against current host state → the new node is picked up → fixed.
|
||||
- A full host reboot restarts the container too (`restart: always`), but as a boot-time race: if the
|
||||
container starts before the USB subsystem finishes settling, or the printer re-enumerated some
|
||||
time *before* the reboot and Docker doesn't necessarily redo an already-satisfied bind-mount
|
||||
target on a policy-driven restart, the container can come back up still bound to the pre-incident
|
||||
view. This matches the exact reported asymmetry (reboot doesn't fix it; explicit restart does).
|
||||
|
||||
**Not yet confirmed on hardware** — this is the leading theory, not a verified root cause. To
|
||||
confirm at the next occurrence, BEFORE restarting anything:
|
||||
```bash
|
||||
# host:
|
||||
ls -la /dev/usb/ && stat /dev/usb/lp1
|
||||
# container:
|
||||
docker exec server ls -la /dev/usb/ && docker exec server stat /dev/usb/lp1
|
||||
```
|
||||
A major:minor or inode mismatch between host and container is the smoking gun. Also worth
|
||||
capturing on the lab RONGTA (different printer, but same cover-open mechanism is plausible):
|
||||
`watch -n1 lsusb` + `sudo dmesg -w | grep -i -E 'usb|disconnect'` while cycling the cover, to see
|
||||
whether the Bus/Device number changes.
|
||||
|
||||
**Candidate fixes, not yet implemented** (ranked cheapest-to-most-invasive):
|
||||
1. A host-side watchdog/udev rule that detects re-enumeration of this printer (match vendor:product
|
||||
ID) and runs `docker restart server` automatically — turns the manual SSH fix into a self-healing
|
||||
one without touching app code.
|
||||
2. Same idea but event-driven via a udev rule or systemd path unit watching `/dev/usb`, rather than
|
||||
polling.
|
||||
3. Switch the compose device wiring from the directory bind-mount to a specific `--device=` cgroup
|
||||
passthrough + a udev rule pinning a stable symlink name — reintroduces the renumbering fragility
|
||||
the directory bind-mount was chosen to avoid, so only worth doing alongside (1)/(2), not instead.
|
||||
|
||||
**Open sub-question — printer identity.** The park-buzi unit shows as "Generic (unknown)" in the
|
||||
app; not yet identified by vendor/product ID. Lab reproduction uses a **RONGTA** unit instead (not
|
||||
the same hardware), so the lab cannot currently reproduce the park-buzi symptom directly — only
|
||||
validate the general re-enumeration mechanism. Commands to identify the real park-buzi printer next
|
||||
time it's reachable via SSH: `lsusb`, `udevadm info -q property -n /dev/usb/lp1`, `udevadm info -a
|
||||
-n /dev/usb/lp1`. This mirrors the same discovery gap already noted above under "Device discovery"
|
||||
(sysfs `ieee1284_id` enrichment) — once identified, fold the model into that mechanism's coverage.
|
||||
|
||||
Related: [[rongta-printer]], [[printer-status-monitoring]], [[printer-roles-failover]],
|
||||
[[appliance-provisioning]], [[network-isolation]], [[technology-stack]].
|
||||
|
||||
+2
-2
@@ -58,7 +58,7 @@ Counts: 4 sources · 19 entities · 47 concepts · 8 decision records.
|
||||
- [[hardware-signer-options]] — where the ledger signing key should live (TPM interim → USB-HSM target; ATECC608 upcoming, not on-site) so a host-owner can't forge the chain.
|
||||
- [[reconciliation]] — the real anti-fraud control; what remote sync actually is.
|
||||
- [[disk-os-hardening]] — the *why* of host hardening: LUKS FDE + TPM-sealed auto-unlock (PCR 7) + Secure Boot + GRUB edit-lock + unprivileged operator + firmware/dbx lockdown; secondary control (reconciliation is the main event). Commands → [[appliance-provisioning]].
|
||||
- [[backup-recovery]] — admin-driven encrypted full-DB backup (local/SMB/SFTP) + DR; signing key escrowed & decoupled from TPM so the ledger survives total hardware loss; restore is admin-only.
|
||||
- [[backup-recovery]] — admin-driven encrypted full-DB backup (local/SMB/SFTP) + DR; signing key escrowed & decoupled from TPM so the ledger survives total hardware loss; restore is admin-only; last-success/error status + schedule are now restart-durable (migration 0025, fixed a "shows Never despite valid backups" bug).
|
||||
|
||||
## Concepts — device architecture & safety
|
||||
- [[device-adapter-pattern]] — business logic talks to interfaces; swap hardware → new adapter.
|
||||
@@ -69,7 +69,7 @@ Counts: 4 sources · 19 entities · 47 concepts · 8 decision records.
|
||||
- [[barrier-not-a-door]] — never timed-close a barrier; safety lives in barrier firmware.
|
||||
- [[printer-roles-failover]] — ≥2 printers by role; entry ticket falls back outside→booth.
|
||||
- [[printer-status-monitoring]] — live poll of paper/cover/cutter/offline via the device's status page; SSE to the booth UI.
|
||||
- [[printer-usb-transport]] — ESC/POS drivers drive TCP (9100) OR local USB (/dev/usb/lp0) behind one render layer; USB = usblp char device, reachability-only status; provisioning open (oq#14).
|
||||
- [[printer-usb-transport]] — ESC/POS drivers drive TCP (9100) OR local USB (/dev/usb/lp0) behind one render layer; USB = usblp char device, reachability-only status; provisioning open (oq#14); park-buzi cover-open-wedges-USB-status bug (docker restart-only fix) under investigation.
|
||||
- [[device-status-monitoring]] — unified live status across ALL device categories (healthCheck + printer readStatus) → the booth footer over /api/ws.
|
||||
- [[trust-boundary]] — the core fork: network vs. device; auditable vs. unforgeable.
|
||||
- [[fail-state-safety]] — entry fails closed, exit fails open; manual override; watchdog.
|
||||
|
||||
+37
@@ -2673,3 +2673,40 @@ the DS-2CD1047G3H-LIU units rather than carry an RTSP/ffmpeg workaround dependen
|
||||
DS-2CD1043G2-LIU (no such bug, ISAPI main-stream snapshot works natively) is the reference model
|
||||
going forward. RTSP main-stream capture remains documented as a proven, viable fallback if a G3H
|
||||
camera is ever unavoidable, but is not being built. Full sweep table + reasoning on [[lpr-camera]].
|
||||
|
||||
## [2026-08-30] update | Booth USB printer cover-open bug: leading theory is a stale container bind-mount, not a stale app-layer handle
|
||||
|
||||
Live troubleshooting request (park-buzi): opening the printer's paper-roll cover reliably wedges its
|
||||
status to offline/faulty, surviving a full appliance reboot; only `docker restart server` clears it.
|
||||
Traced `sendRawUsb`/`probeUsb` end-to-end in `printer-escpos.ts` plus both poll loops
|
||||
(`device-monitor.ts`, `printer-monitor.ts`): every print AND every poll does a fresh
|
||||
open→write/probe→close with no persistent fd/socket/driver instance anywhere — ruling out a naive
|
||||
"stale Node handle" explanation. Leading hypothesis instead: the cover-open microswitch cuts power
|
||||
to the printer's USB interface board, causing a real bus re-enumeration; the container's directory
|
||||
bind-mount of `/dev/usb` (chosen specifically to survive `lpN` renumbering) can retain a stale view
|
||||
of the old device node until the container's mount namespace is recreated — which `docker restart`
|
||||
does and a policy-driven reboot-time restart may not (boot-order race). Not yet confirmed on
|
||||
hardware (host-vs-container `stat`/inode comparison at the next occurrence is the next step); lab
|
||||
repro is blocked because the lab has a RONGTA, not the park-buzi unit's actual (still unidentified,
|
||||
"Generic (unknown)") model. Full writeup, confirmation commands, and candidate fixes on
|
||||
[[printer-usb-transport]].
|
||||
|
||||
## [2026-08-30] update | Backup status "Never" despite valid rotating backups — restart amnesia in BackupService, fixed
|
||||
|
||||
Admin noticed park-buzi's Backup screen showed "last successful backup: Never" despite 7 real,
|
||||
correctly-rotating encrypted backup files on disk, plus a 2-day gap since the last file. Traced
|
||||
both symptoms to the same cause: `BackupService` tracked last-success/last-error as PLAIN
|
||||
IN-PROCESS FIELDS (never written to the DB), and the daily schedule was a `setInterval(...,24h)`
|
||||
measured from PROCESS START, not wall-clock time since the last real backup — so any server
|
||||
restart (routine under `restart: always`: deploy/crash/OOM/host reboot) simultaneously wiped the
|
||||
visible status back to "Never" and reset the 24h countdown, independent of the actual
|
||||
file-writing/retention engine (`backup.ts`), which was working correctly the whole time and
|
||||
explains why files existed on disk despite the UI's contradictory-seeming status. Fix: four new
|
||||
nullable `site_config` columns (migration `0025_backup_last_status.sql`) persist last-success/
|
||||
error there instead of in memory; `BackupService.status()` reads them fresh each call so a new
|
||||
instance (= a restart) sees the prior instance's outcome; a new `isDue()` method computes
|
||||
schedule-due-ness from the persisted last-success timestamp; `server.ts`'s scheduler is now a
|
||||
15-minute poll gated by `isDue()` instead of a 24h `setInterval`, making the real cadence immune
|
||||
to restart timing. New test file `backup-service.test.ts` (6 tests) covers restart-durability and
|
||||
`isDue()` directly; full existing suite (319 tests) still green. No API/UI contract change. Not
|
||||
yet committed (holding per instruction). Full writeup on [[backup-recovery]].
|
||||
|
||||
Reference in New Issue
Block a user