From 7c3b10eb2faf7ae5836c482970ad16ba96c905f2 Mon Sep 17 00:00:00 2001 From: "Luis Guzman (AppDevForAll)" Date: Mon, 5 Oct 2026 00:17:08 -0600 Subject: [PATCH 1/2] K2GO-451 fix(dashboard): rebuild status self-heals a crashed (dead-pid) running state A rebuild whose script is killed mid-build (box restart / power loss) cannot write a terminal status, so /rebuild/status reported 'running' forever and a new rebuild was refused with 409. The status read now checks pid liveness and heals a dead-pid 'running' to 'error' (clearing lock/phase/pid). The dead-pid check and state reset are shared with the cancel handler. dash-node 1.4.0. --- static/dashboard/CHANGELOG.md | 1 + static/dashboard/package.json | 2 +- static/dashboard/routes.ts | 62 ++++++++++++++++++++++++----------- 3 files changed, 44 insertions(+), 21 deletions(-) diff --git a/static/dashboard/CHANGELOG.md b/static/dashboard/CHANGELOG.md index cdde5af7..32c3e435 100644 --- a/static/dashboard/CHANGELOG.md +++ b/static/dashboard/CHANGELOG.md @@ -4,6 +4,7 @@ One line per version, newest first. Every REST-facing change bumps the version i (the app surfaces it via `/system/dashboard/update-check` and the "Update available" pill), so this file is the human record of what each bump enables. Keep entries short: `version - change (TICKET)`. +- **1.4.0** - Rebuild status self-heals a crashed run (K2GO-451). `GET /rebuild/status` now checks whether the recorded rebuild pid is still alive; a `running` status left behind by a script that was killed mid-build (box restart / power loss) is healed to `error` and the lock/phase/pid files are cleared, so the app sees a normal failure instead of polling a phantom `running` forever and a new rebuild is no longer refused with 409. The dead-pid check and state reset are shared with the cancel handler (one source). A `running` with no pid yet (a just started run) is left untouched. Localhost-only. (K2GO-451) - **1.3.13** - Forgejo repo refresh on the durable job engine (K2GO-443). `forgejo` is now a job type: `POST /forgejo/download` plus `GET /forgejo/jobs/:id` (structured `{phase, percent, detail}`) and retry/cancel over the generic `/:type/*` surface. The runner (`sockets/forgejo.exec.ts`) wraps the existing box orchestration (`static/forgejo/orchestration` -> `refresh_forgejo`) and reports per-repo progress (repo N of M + the current repo name) parsed from a new `K2GO_PROGRESS` marker the orchestration emits. Forgejo is a git operation (fetch + fast-forward/merge + authenticated push per seeded example repo), not a file download, so there is NO aria2 and NO pause/resume; retry re-runs the idempotent refresh. The seed (install) path is unchanged. The older `POST /forgejo/refresh` (wrapper) stays for now. Localhost-only. (K2GO-443) - **1.3.12** - Add-ons gallery download on the durable job engine (K2GO-443). `code-addons` is now a job type: `POST /code-addons/download` plus `GET /code-addons/jobs/:id` (structured `{phase, percent, speed, detail}`) and pause/resume/retry/cancel over the generic `/:type/*` surface, like build-assets. The runner (`sockets/code_addons.exec.ts`) downloads only the heavy add-on binaries (.cgp + source tarballs) with aria2c (resilient: `--continue` resume, survives a network change) using the shared `downloadWithAria2` helper; the mirror stages the small files (shell, catalog, icons, pages) with its Cloudflare clean + catalog base rewrite and prints the aria2 input for the heavy ones (`mirror_addons.py --print-aria2-input`), then verifies them (`--finalize-only`), and the runner swaps the staged tree in atomically. The older `POST /addons/refresh` (wrapper) stays for now. Localhost-only. (K2GO-443) - **1.3.11** - Build-assets download on the durable job engine (K2GO-443). `code-assets` is now a job type: `POST /code-assets/download` plus `GET /code-assets/jobs/:id` (structured `{phase, percent, speed, detail}`) and pause/resume/retry/cancel over the generic `/:type/*` surface, like kiwix/maps. The runner (`sockets/code_assets.exec.ts`) downloads the build assets with aria2c (resilient: `--continue` resume, survives a full interface loss via the outer retry loop) using the shared `downloadWithAria2` helper, then the mirror verifies each file against its published `.md5` and writes the browse page (`mirror_code_assets.py --finalize-only`), and the runner swaps the staged tree in atomically. The older `POST /code-assets/refresh` (wrapper) stays for now. Localhost-only. (K2GO-443) diff --git a/static/dashboard/package.json b/static/dashboard/package.json index 551a0210..00aa133c 100644 --- a/static/dashboard/package.json +++ b/static/dashboard/package.json @@ -1,6 +1,6 @@ { "name": "dashboard-console", - "version": "1.3.13", + "version": "1.4.0", "description": "", "main": "index.js", "scripts": { diff --git a/static/dashboard/routes.ts b/static/dashboard/routes.ts index 90c90343..0556a400 100644 --- a/static/dashboard/routes.ts +++ b/static/dashboard/routes.ts @@ -295,10 +295,42 @@ apiRouter.get('/system/disk-guard/firehose', (_req: Request, res: Response): voi }); }); +// K2GO-451: one answer to "is the recorded rebuild process still alive?", shared by the status read and +// the cancel handler so the dead-pid check lives in one place. 'none' = no pid recorded yet (a just +// started run; never stale), 'dead' = the recorded pid is gone (the script was killed, e.g. a box restart +// mid-build), 'alive' = /proc/ exists (ours, or a recycled unrelated pid: cmdline tells which). +// /proc//cmdline is NUL-separated; a substring match on the script name is enough. +function rebuildPidLiveness(): { pid: number; state: 'none' | 'dead' | 'alive'; cmdline: string } { + let pid = 0; + try { pid = parseInt(fs.readFileSync(REBUILD_PID_FILE, 'utf8').trim(), 10); } catch { /* no pid yet */ } + if (!Number.isFinite(pid) || pid <= 1) return { pid: 0, state: 'none', cmdline: '' }; + let cmdline = ''; + try { cmdline = fs.readFileSync(`/proc/${pid}/cmdline`, 'utf8'); } catch { /* process gone / no proc */ } + if (cmdline === '' && !fs.existsSync(`/proc/${pid}`)) return { pid, state: 'dead', cmdline: '' }; + return { pid, state: 'alive', cmdline }; +} + +// K2GO-451: reset the rebuild run's state files to a terminal value. 'idle' after a user cancel; 'error' +// when a crashed run (dead pid) is healed, so the app sees a normal failure instead of endless 'running'. +function clearRebuildState(status: 'idle' | 'error'): void { + try { fs.writeFileSync(REBUILD_STATUS_FILE, status); } catch { /* best effort */ } + try { fs.rmdirSync(REBUILD_LOCK_DIR); } catch { /* maybe already gone */ } + try { fs.unlinkSync(REBUILD_PHASE_FILE); } catch { /* maybe already gone */ } + try { fs.unlinkSync(REBUILD_PID_FILE); } catch { /* maybe already gone */ } +} + // Current rebuild state: idle | running | done | error (read from the status file the script writes). apiRouter.get('/system/dashboard/rebuild/status', (_req: Request, res: Response): void => { let state = 'idle'; try { state = (fs.readFileSync(REBUILD_STATUS_FILE, 'utf8').trim() || 'idle'); } catch { /* no file yet */ } + // K2GO-451: a rebuild whose script was killed mid-run (box restart / power loss) cannot write a + // terminal status, so this file stays 'running' with a dead pid and nothing clears it. Heal it to + // 'error' on read, so the app sees a normal failure instead of polling a phantom forever (and a new + // rebuild is no longer refused with 409). A 'running' with no pid yet is a just-started run: leave it. + if (state === 'running' && rebuildPidLiveness().state === 'dead') { + clearRebuildState('error'); + state = 'error'; + } res.json({ state }); }); @@ -641,39 +673,29 @@ apiRouter.post('/system/dashboard/rebuild/cancel', (_req: Request, res: Response let phase = 'building'; try { phase = fs.readFileSync(REBUILD_PHASE_FILE, 'utf8').trim() || 'building'; } catch { /* default building */ } if (phase === 'promoting') { res.status(409).json({ error: 'promoting', promoting: true, cancelled: false }); return; } - let pid = 0; - try { pid = parseInt(fs.readFileSync(REBUILD_PID_FILE, 'utf8').trim(), 10); } catch { /* no pid yet */ } - if (!Number.isFinite(pid) || pid <= 1) { + // K2GO-451: the dead-pid check now lives in rebuildPidLiveness() (shared with the status read). + const live = rebuildPidLiveness(); + if (live.state === 'none') { // The run has just started (status is 'running') but hasn't recorded its session pid yet, so we // can't signal it. Do NOT claim success or touch the status — the app can retry in a moment. res.status(409).json({ error: 'cancel not ready', cancelled: false }); return; } - // Verify the pid is actually OUR rebuild script before signaling — never SIGTERM a recycled/unrelated - // pid. /proc//cmdline is NUL-separated; a substring match on the script name is enough. - let cmdline = ''; - try { cmdline = fs.readFileSync(`/proc/${pid}/cmdline`, 'utf8'); } catch { /* process gone / no proc */ } - const healStale = (): void => { - try { fs.writeFileSync(REBUILD_STATUS_FILE, 'idle'); } catch { /* best effort */ } - try { fs.rmdirSync(REBUILD_LOCK_DIR); } catch { /* maybe already gone */ } - try { fs.unlinkSync(REBUILD_PHASE_FILE); } catch { /* maybe already gone */ } - try { fs.unlinkSync(REBUILD_PID_FILE); } catch { /* maybe already gone */ } - }; - if (cmdline === '' && !fs.existsSync(`/proc/${pid}`)) { + if (live.state === 'dead') { // The run already exited but left stale state (status 'running' with a dead pid) — heal it and // report success; there is nothing left to stop. - healStale(); + clearRebuildState('idle'); res.status(200).json({ ok: true, cancelled: true }); return; } - if (!cmdline.includes('rebuild-dashboard.sh')) { + if (!live.cmdline.includes('rebuild-dashboard.sh')) { // The pid is alive but was recycled by an unrelated process — refuse to signal it. res.status(409).json({ error: 'stale pid', cancelled: false }); return; } try { // Signal the whole detached session group; the script's on_cancel/EXIT trap sets status idle and - // purges staging + lock/phase/pid. We also write idle + best-effort rmdir here as belt-and-suspenders. - try { process.kill(-pid, 'SIGTERM'); } catch { /* group already gone */ } - try { process.kill(pid, 'SIGTERM'); } catch { /* leader already gone */ } - healStale(); + // purges staging + lock/phase/pid. We also clear here as belt-and-suspenders. + try { process.kill(-live.pid, 'SIGTERM'); } catch { /* group already gone */ } + try { process.kill(live.pid, 'SIGTERM'); } catch { /* leader already gone */ } + clearRebuildState('idle'); res.status(200).json({ ok: true, cancelled: true }); } catch (e: any) { res.status(500).json({ error: e?.message || 'cancel failed', cancelled: false }); From a7b3033f1ee3d43f1ff57d61b674987547be9c29 Mon Sep 17 00:00:00 2001 From: "Luis Guzman (AppDevForAll)" Date: Mon, 5 Oct 2026 00:59:07 -0600 Subject: [PATCH 2/2] K2GO-451 fix(dashboard): heal a crashed rebuild with no pid recorded; unit-test the decision Review follow-up. The status read and the new-rebuild guard now both go through readRebuildStatus(), which also heals a 'running' that has outlived a short grace with no pid ever recorded (the script died before, or without, writing one), not just a dead recorded pid. The heal decision moves to the pure, unit-tested resolveRebuildState. A running within the grace (just started) is left untouched. --- static/dashboard/CHANGELOG.md | 2 +- static/dashboard/package.json | 2 +- static/dashboard/routes.ts | 46 +++++++++++++------ .../dashboard/sockets/rebuild-status.test.ts | 36 +++++++++++++++ static/dashboard/sockets/rebuild-status.ts | 37 +++++++++++++++ 5 files changed, 106 insertions(+), 17 deletions(-) create mode 100644 static/dashboard/sockets/rebuild-status.test.ts create mode 100644 static/dashboard/sockets/rebuild-status.ts diff --git a/static/dashboard/CHANGELOG.md b/static/dashboard/CHANGELOG.md index 32c3e435..01af1709 100644 --- a/static/dashboard/CHANGELOG.md +++ b/static/dashboard/CHANGELOG.md @@ -4,7 +4,7 @@ One line per version, newest first. Every REST-facing change bumps the version i (the app surfaces it via `/system/dashboard/update-check` and the "Update available" pill), so this file is the human record of what each bump enables. Keep entries short: `version - change (TICKET)`. -- **1.4.0** - Rebuild status self-heals a crashed run (K2GO-451). `GET /rebuild/status` now checks whether the recorded rebuild pid is still alive; a `running` status left behind by a script that was killed mid-build (box restart / power loss) is healed to `error` and the lock/phase/pid files are cleared, so the app sees a normal failure instead of polling a phantom `running` forever and a new rebuild is no longer refused with 409. The dead-pid check and state reset are shared with the cancel handler (one source). A `running` with no pid yet (a just started run) is left untouched. Localhost-only. (K2GO-451) +- **1.4.0** - Rebuild status self-heals a crashed run (K2GO-451). A `running` status left behind by a script that was killed mid-build (box restart / power loss) is healed to `error` and the lock/phase/pid files are cleared, so the app sees a normal failure instead of polling a phantom `running` forever and a new rebuild is no longer refused with 409. `GET /rebuild/status` and the new-rebuild guard both read through `readRebuildStatus()`, which heals when the recorded pid is dead, or when a `running` has outlived a short grace with no pid ever recorded (the script died before, or without, writing one). A `running` with a live pid, or one still within the grace (a just started run), is left untouched. The dead-pid check and state reset are shared with the cancel handler (one source); the heal decision is the pure, unit-tested `resolveRebuildState`. Localhost-only. (K2GO-451) - **1.3.13** - Forgejo repo refresh on the durable job engine (K2GO-443). `forgejo` is now a job type: `POST /forgejo/download` plus `GET /forgejo/jobs/:id` (structured `{phase, percent, detail}`) and retry/cancel over the generic `/:type/*` surface. The runner (`sockets/forgejo.exec.ts`) wraps the existing box orchestration (`static/forgejo/orchestration` -> `refresh_forgejo`) and reports per-repo progress (repo N of M + the current repo name) parsed from a new `K2GO_PROGRESS` marker the orchestration emits. Forgejo is a git operation (fetch + fast-forward/merge + authenticated push per seeded example repo), not a file download, so there is NO aria2 and NO pause/resume; retry re-runs the idempotent refresh. The seed (install) path is unchanged. The older `POST /forgejo/refresh` (wrapper) stays for now. Localhost-only. (K2GO-443) - **1.3.12** - Add-ons gallery download on the durable job engine (K2GO-443). `code-addons` is now a job type: `POST /code-addons/download` plus `GET /code-addons/jobs/:id` (structured `{phase, percent, speed, detail}`) and pause/resume/retry/cancel over the generic `/:type/*` surface, like build-assets. The runner (`sockets/code_addons.exec.ts`) downloads only the heavy add-on binaries (.cgp + source tarballs) with aria2c (resilient: `--continue` resume, survives a network change) using the shared `downloadWithAria2` helper; the mirror stages the small files (shell, catalog, icons, pages) with its Cloudflare clean + catalog base rewrite and prints the aria2 input for the heavy ones (`mirror_addons.py --print-aria2-input`), then verifies them (`--finalize-only`), and the runner swaps the staged tree in atomically. The older `POST /addons/refresh` (wrapper) stays for now. Localhost-only. (K2GO-443) - **1.3.11** - Build-assets download on the durable job engine (K2GO-443). `code-assets` is now a job type: `POST /code-assets/download` plus `GET /code-assets/jobs/:id` (structured `{phase, percent, speed, detail}`) and pause/resume/retry/cancel over the generic `/:type/*` surface, like kiwix/maps. The runner (`sockets/code_assets.exec.ts`) downloads the build assets with aria2c (resilient: `--continue` resume, survives a full interface loss via the outer retry loop) using the shared `downloadWithAria2` helper, then the mirror verifies each file against its published `.md5` and writes the browse page (`mirror_code_assets.py --finalize-only`), and the runner swaps the staged tree in atomically. The older `POST /code-assets/refresh` (wrapper) stays for now. Localhost-only. (K2GO-443) diff --git a/static/dashboard/package.json b/static/dashboard/package.json index 00aa133c..4a2fc93a 100644 --- a/static/dashboard/package.json +++ b/static/dashboard/package.json @@ -4,7 +4,7 @@ "description": "", "main": "index.js", "scripts": { - "test": "node --require ts-node/register --test sockets/maps.socket.test.ts sockets/rolling-log.test.ts sockets/kolibri.session.test.ts sockets/credentials.test.ts sockets/net-retry.test.ts sockets/services.test.ts sockets/log-rotate.test.ts", + "test": "node --require ts-node/register --test sockets/maps.socket.test.ts sockets/rolling-log.test.ts sockets/kolibri.session.test.ts sockets/credentials.test.ts sockets/net-retry.test.ts sockets/services.test.ts sockets/log-rotate.test.ts sockets/rebuild-status.test.ts", "test:db": "node --require ts-node/register --test sockets/jobs.test.ts", "typecheck": "tsc --noEmit", "build": "tsc", diff --git a/static/dashboard/routes.ts b/static/dashboard/routes.ts index 0556a400..e20ab02b 100644 --- a/static/dashboard/routes.ts +++ b/static/dashboard/routes.ts @@ -24,6 +24,7 @@ import { } from './sockets/credentials'; import { isRestartableService, restartService } from './sockets/services'; import { getFirehoseState } from './sockets/log-rotate'; +import { resolveRebuildState } from './sockets/rebuild-status'; // ADFA-4879: FQR helpers reached from the app (in-app region download/delete instead of the // copy-paste-into-a-terminal flow). tile-extract.py is installed on the box by the upstream maps @@ -319,19 +320,34 @@ function clearRebuildState(status: 'idle' | 'error'): void { try { fs.unlinkSync(REBUILD_PID_FILE); } catch { /* maybe already gone */ } } -// Current rebuild state: idle | running | done | error (read from the status file the script writes). -apiRouter.get('/system/dashboard/rebuild/status', (_req: Request, res: Response): void => { - let state = 'idle'; - try { state = (fs.readFileSync(REBUILD_STATUS_FILE, 'utf8').trim() || 'idle'); } catch { /* no file yet */ } - // K2GO-451: a rebuild whose script was killed mid-run (box restart / power loss) cannot write a - // terminal status, so this file stays 'running' with a dead pid and nothing clears it. Heal it to - // 'error' on read, so the app sees a normal failure instead of polling a phantom forever (and a new - // rebuild is no longer refused with 409). A 'running' with no pid yet is a just-started run: leave it. - if (state === 'running' && rebuildPidLiveness().state === 'dead') { - clearRebuildState('error'); - state = 'error'; +// K2GO-451: a live rebuild records its pid within ~1s of taking the lock, so a 'running' with no pid +// recorded past this grace means the script died before (or without) writing one (or never started). +const REBUILD_PID_GRACE_MS = 15_000; + +// K2GO-451: read the status file and self-heal a 'running' that no live rebuild backs (the script was +// killed mid-run by a box restart / power loss, so it never wrote a terminal status). Shared by the +// status endpoint and the new-rebuild guard, so a crashed run neither reports a phantom 'running' nor +// blocks a fresh rebuild with 409. The decision is the pure resolveRebuildState; this only does the IO. +function readRebuildStatus(): string { + let fileState = 'idle'; + try { fileState = fs.readFileSync(REBUILD_STATUS_FILE, 'utf8').trim() || 'idle'; } catch { return 'idle'; } + if (fileState !== 'running') return fileState; + const live = rebuildPidLiveness(); + let ageMs = 0; + if (live.state === 'none') { + // Age the 'running' from the status file's mtime (written at run start); only needed to tell a + // just-started run (no pid yet) from one whose script died before recording a pid. + try { ageMs = Date.now() - fs.statSync(REBUILD_STATUS_FILE).mtimeMs; } catch { ageMs = 0; } } - res.json({ state }); + const r = resolveRebuildState(fileState, live.state, ageMs, REBUILD_PID_GRACE_MS); + if (r.heal) clearRebuildState('error'); + return r.state; +} + +// Current rebuild state: idle | running | done | error (read from the status file the script writes; +// a stale 'running' left by a crashed run is healed to 'error' here). +apiRouter.get('/system/dashboard/rebuild/status', (_req: Request, res: Response): void => { + res.json({ state: readRebuildStatus() }); }); // ADFA-5339: read-only tail of the rebuild log, for the card's expandable Details. Returns the last @@ -637,9 +653,9 @@ apiRouter.post('/code-assets/refresh/cancel', (_req: Request, res: Response): vo // so it matches the new source. It never touches the reported version; a site failure is logged and // does NOT fail the (already-verified) core update. See ADFA-5339 §semantics. apiRouter.post('/system/dashboard/rebuild', (req: Request, res: Response): void => { - let running = false; - try { running = fs.readFileSync(REBUILD_STATUS_FILE, 'utf8').trim() === 'running'; } catch { /* none */ } - if (running) { res.status(409).json({ error: 'a rebuild is already running' }); return; } + // K2GO-451: readRebuildStatus heals a crashed run's stale 'running' first, so a dead rebuild does not + // block a fresh one with a 409 the user cannot clear. + if (readRebuildStatus() === 'running') { res.status(409).json({ error: 'a rebuild is already running' }); return; } if (!fs.existsSync(REBUILD_SCRIPT)) { res.status(500).json({ error: 'rebuild script not found' }); return; } const updateSite = (req.body as { site?: unknown })?.site === true; try { diff --git a/static/dashboard/sockets/rebuild-status.test.ts b/static/dashboard/sockets/rebuild-status.test.ts new file mode 100644 index 00000000..468f0a32 --- /dev/null +++ b/static/dashboard/sockets/rebuild-status.test.ts @@ -0,0 +1,36 @@ +/// +import test from 'node:test'; +import assert from 'node:assert/strict'; +import { resolveRebuildState } from './rebuild-status'; + +const GRACE = 15_000; + +// --- resolveRebuildState: the pure heal decision for the rebuild status (K2GO-451) ------------- + +test('resolveRebuildState: non-running states pass through untouched, no heal', () => { + for (const s of ['idle', 'done', 'error']) { + assert.deepEqual(resolveRebuildState(s, 'none', 0, GRACE), { state: s, heal: false }); + // the pid liveness and age are irrelevant once the file is not 'running' + assert.deepEqual(resolveRebuildState(s, 'dead', 1_000_000, GRACE), { state: s, heal: false }); + } +}); + +test('resolveRebuildState: running with a dead recorded pid heals to error', () => { + assert.deepEqual(resolveRebuildState('running', 'dead', 0, GRACE), { state: 'error', heal: true }); +}); + +test('resolveRebuildState: running with an alive pid stays running, no heal', () => { + // a real rebuild, or a recycled unrelated pid: never healed on the read path + assert.deepEqual(resolveRebuildState('running', 'alive', 1_000_000, GRACE), { state: 'running', heal: false }); +}); + +test('resolveRebuildState: running with no pid yet, within the grace, stays running', () => { + // a just-started run that has not written its pid: must not be false-healed + assert.deepEqual(resolveRebuildState('running', 'none', 0, GRACE), { state: 'running', heal: false }); + assert.deepEqual(resolveRebuildState('running', 'none', GRACE, GRACE), { state: 'running', heal: false }); // boundary: not > grace +}); + +test('resolveRebuildState: running with no pid past the grace heals to error', () => { + // the script died before (or without) recording a pid, or never started + assert.deepEqual(resolveRebuildState('running', 'none', GRACE + 1, GRACE), { state: 'error', heal: true }); +}); diff --git a/static/dashboard/sockets/rebuild-status.ts b/static/dashboard/sockets/rebuild-status.ts new file mode 100644 index 00000000..72e4fa40 --- /dev/null +++ b/static/dashboard/sockets/rebuild-status.ts @@ -0,0 +1,37 @@ +// K2GO-451: the pure decision for what a reader of the dash-node rebuild status should report, and +// whether it must heal a stale "running". Kept free of fs / process so it is unit-tested off-box; the +// thin wrapper in routes.ts feeds it the status file's value, the recorded pid's liveness, and how long +// "running" has been set. A "running" status is only trustworthy while a live rebuild backs it: +// - pid "dead" : the script was killed (box restart / power loss) -> heal to "error". +// - pid "none" past the grace : the script died before (or without) recording its pid -> heal to "error". +// - pid "none" within the grace : a just started run that has not written its pid yet -> leave "running". +// - pid "alive" : a real rebuild (or a recycled pid) -> leave "running". +// Any non-"running" file value passes through untouched. Healing targets "error" (a crash is a failure, +// and the app's poller treats "error" as terminal; "idle" would read as still running and loop). + +export type PidState = 'none' | 'dead' | 'alive'; + +export interface RebuildResolution { + /** The effective status a reader should report. */ + state: string; + /** True when the caller must clear the stale run's state files (status -> "error", drop lock/phase/pid). */ + heal: boolean; +} + +export function resolveRebuildState( + fileState: string, + pidState: PidState, + runningAgeMs: number, + graceMs: number, +): RebuildResolution { + if (fileState !== 'running') { + return { state: fileState, heal: false }; + } + if (pidState === 'dead') { + return { state: 'error', heal: true }; + } + if (pidState === 'none' && runningAgeMs > graceMs) { + return { state: 'error', heal: true }; + } + return { state: 'running', heal: false }; +}