From 7e68e18a17f3ea3b2841d5a057a9af0d96c6cc33 Mon Sep 17 00:00:00 2001 From: ScreenTinker Date: Tue, 30 Jun 2026 22:08:55 -0500 Subject: [PATCH] =?UTF-8?q?test(#146)=20P2.6:=20boot=20health=20during=20a?= =?UTF-8?q?=20large=20startup=20trim=20=E2=80=94=20confirmed?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Booting against a pre-bloated 300k-row device_status_log, /api/status answers in <3s while the table is still large (chunked startup prune trickling in the background), and the backlog drains to the cap with the server responsive throughout. The old whole-table sort froze boot ~40s. Fallout doc gets the P2 findings section. Co-Authored-By: Claude Opus 4.8 (1M context) --- docs/146-hardening-fallout.md | 13 +++++++ server/test/boot-health.test.js | 65 +++++++++++++++++++++++++++++++++ 2 files changed, 78 insertions(+) create mode 100644 server/test/boot-health.test.js diff --git a/docs/146-hardening-fallout.md b/docs/146-hardening-fallout.md index 4ce9add..8357e5c 100644 --- a/docs/146-hardening-fallout.md +++ b/docs/146-hardening-fallout.md @@ -90,6 +90,19 @@ change or bisect. All read at process start. Note the startup prune is intentionally NEVER band-gated (it must clear a boot-time backlog); `MAINTENANCE_BAND_GATE_ENABLED` only affects the interval run. +## P2 findings (investigated; measured) +- **Playlist build under mass reconnect — MITIGATED, no change.** `buildPlaylistPayload` + is synchronous (indexed SELECTs + one `JSON.parse` of the published snapshot). Measured + with a 200-item snapshot: **avg 0.078ms, max 0.70ms per call**; a 230-device fleet-wide + reconnect is ~18ms of CPU **spread across 230 separate socket.io handler invocations** + (the loop yields between them), never one block. Per-call is the real unit and is far + under the 50ms invariant. (`test/playlist-build-cost.test.js`.) +- **Boot health during a large startup trim — CONFIRMED.** Booting against a pre-bloated + 300k-row `device_status_log`, `/api/status` answers in **<3s** while the table is still + large (prune trickling), and the backlog drains to the cap in the background with the + server responsive throughout. The old whole-table sort froze boot ~40s. This is the + point of the async/chunked, un-gated startup prune. (`test/boot-health.test.js`.) + ## Interlock note Item A ends the prune-induced restart loop; Item B's in-memory flap state now persists long enough to bite (it used to be wiped every ~40s by the restart). The two are a pair: diff --git a/server/test/boot-health.test.js b/server/test/boot-health.test.js new file mode 100644 index 0000000..52cc4a3 --- /dev/null +++ b/server/test/boot-health.test.js @@ -0,0 +1,65 @@ +'use strict'; + +// #146 P2.6 — a deploy against a PRE-BLOATED device_status_log must still bind + serve +// /api/status quickly, with the chunked startup prune trickling in the background (the +// whole point of the async/chunked startup prune — the old whole-table sort froze boot +// -> healthcheck fail -> restart loop). Seed a big backlog, boot, assert /api/status +// answers fast WHILE the table is still large, then confirm the backlog drains. + +const { test } = require('node:test'); +const assert = require('node:assert/strict'); +const { spawn } = require('node:child_process'); +const path = require('node:path'); +const os = require('node:os'); +const fs = require('node:fs'); +const crypto = require('node:crypto'); +const Database = require('better-sqlite3'); + +const PORT = 3995; +const BASE = `http://127.0.0.1:${PORT}`; +const DATA_DIR = path.join(os.tmpdir(), 'st-boot-' + crypto.randomBytes(4).toString('hex')); +const DBPATH = path.join(DATA_DIR, 'db', 'remote_display.db'); + +test('boots + serves /api/status quickly against a pre-bloated table; prune drains in background', async () => { + // 1) Create + migrate the DB in a throwaway boot, then seed a large backlog. + { + const p = spawn('node', ['server.js'], { cwd: path.join(__dirname, '..'), env: { ...process.env, DATA_DIR, SELF_HOSTED: 'true', PORT: '3894', NODE_ENV: 'test' }, stdio: 'ignore' }); + for (let i = 0; i < 60; i++) { try { const r = await fetch('http://127.0.0.1:3894/api/status'); if (r.ok) break; } catch { /* */ } await new Promise(r => setTimeout(r, 200)); } + p.kill('SIGKILL'); + await new Promise(r => setTimeout(r, 300)); + } + const seed = new Database(DBPATH); + const ins = seed.prepare('INSERT INTO device_status_log (device_id, status, timestamp) VALUES (?, ?, ?)'); + const now = Math.floor(Date.now() / 1000); + seed.transaction(() => { for (let i = 0; i < 300000; i++) ins.run('d' + (i % 3), 'online', now); })(); + assert.equal(seed.prepare('SELECT COUNT(*) c FROM device_status_log').get().c, 300000, 'seeded 300k backlog'); + seed.close(); + + // 2) Boot for real against the bloated table; time to first /api/status OK. + const proc = spawn('node', ['server.js'], { cwd: path.join(__dirname, '..'), env: { ...process.env, DATA_DIR, SELF_HOSTED: 'true', PORT: String(PORT), NODE_ENV: 'test', STATUS_LOG_MAX_ROWS_PER_DEVICE: '500' }, stdio: ['ignore', fs.openSync(path.join(os.tmpdir(), 'st-boot.log'), 'w'), 'inherit'] }); + try { + const t0 = Date.now(); + let up = false; + for (let i = 0; i < 60; i++) { try { const r = await fetch(BASE + '/api/status'); if (r.ok) { up = true; break; } } catch { /* */ } await new Promise(r => setTimeout(r, 100)); } + const bootMs = Date.now() - t0; + assert.ok(up, 'server bound and served /api/status'); + assert.ok(bootMs < 3000, `/api/status answered in ${bootMs}ms — NOT blocked by the 300k prune (old sort froze ~40s)`); + + // At first-serve the prune is still trickling: the table should still be large. + const ro1 = new Database(DBPATH, { readonly: true }); + const atBoot = ro1.prepare('SELECT COUNT(*) c FROM device_status_log').get().c; ro1.close(); + assert.ok(atBoot > 1500, `prune runs in background — table still large at first serve (${atBoot})`); + + // Give the chunked startup prune time to drain, staying responsive throughout. + for (let i = 0; i < 40; i++) { + const r = await fetch(BASE + '/api/status'); assert.ok(r.ok, 'stays responsive while pruning'); + const ro = new Database(DBPATH, { readonly: true }); + const c = ro.prepare('SELECT COUNT(*) c FROM device_status_log').get().c; ro.close(); + if (c <= 1500) break; + await new Promise(r => setTimeout(r, 250)); + } + const ro2 = new Database(DBPATH, { readonly: true }); + const drained = ro2.prepare('SELECT COUNT(*) c FROM device_status_log').get().c; ro2.close(); + assert.equal(drained, 1500, '3 devices x 500 cap — backlog fully drained by the background startup prune'); + } finally { proc.kill('SIGKILL'); } +});