mirror of
https://github.com/screentinker/screentinker.git
synced 2026-08-14 14:23:14 -06:00
The death-spiral amplifier: pruneStatusLog ran a whole-table ROW_NUMBER() sort, 40-48s synchronous on the 1.1M-row incident table, freezing boot -> healthcheck fail -> restart loop. - lib/chunked-prune.js: shared chunkedDelete (rowid IN (SELECT ... LIMIT ?) since better-sqlite3 has no DELETE...LIMIT) — bounded batch + setImmediate yield between batches, optional band-gate. Core invariant: no sync op blocks >~50ms ever. - pruneStatusLog: rewritten per-device via a loose index-scan seek (WHERE device_id > ? ORDER BY device_id LIMIT 1 — O(log n) each), retention + newest-cap trimmed in bounded batches, async, re-entrancy-guarded, band-gated on the interval / un-gated + fire-and-forget at startup so a bloated table self-heals on deploy WITHOUT freezing boot. - heartbeat.js: maintenance moved off the interval body into async band-gated re-entrant runMaintenance(); play_logs + provisioning prunes chunked; offline-marking stays synchronous. - pruneTelemetry: bounded single statement (OFFSET 6000 LIMIT batch), stays sync. - idx_devices_provisioning so the provisioning prune batch subquery is an index range. Tests: correctness (per-device cap + retention, independent devices), 300k-row backlog trims in many batches with max event-loop gap <250ms, band-gate no-op while critical + startup runs regardless, re-entrancy (concurrent -> once). Existing prune tests updated to await. Suite 247/247. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
142 lines
6 KiB
JavaScript
142 lines
6 KiB
JavaScript
const { db, pruneStatusLog } = require('../db/database');
|
|
const config = require('../config');
|
|
const { deviceRoom, emitToWorkspace } = require('../lib/socket-rooms');
|
|
const statusLogWriter = require('../lib/status-log-writer');
|
|
const { chunkedDelete, currentBand } = require('../lib/chunked-prune'); // #146 non-blocking sweeps
|
|
|
|
// Track connected device sockets: deviceId -> { socketId, lastHeartbeat }
|
|
const deviceConnections = new Map();
|
|
|
|
function startHeartbeatChecker(io) {
|
|
// #146: startup sweep is chunked + async + fire-and-forget + NOT band-gated, so a
|
|
// bloated device_status_log self-heals on next deploy WITHOUT freezing boot (the old
|
|
// whole-table sort froze boot 40-48s -> healthcheck fail -> restart loop). It
|
|
// trickles in bounded batches while the server comes up and serves.
|
|
pruneStatusLog({ bandGate: false }).catch(() => {});
|
|
|
|
// #146: start the batched device_status_log flush loop.
|
|
statusLogWriter.start();
|
|
|
|
const deviceNs = io.of('/device');
|
|
|
|
setInterval(() => {
|
|
const now = Date.now();
|
|
const dashboardNs = io.of('/dashboard');
|
|
|
|
// Check database for devices that should be offline
|
|
const onlineDevices = db.prepare("SELECT id, last_heartbeat FROM devices WHERE status = 'online'").all();
|
|
|
|
for (const device of onlineDevices) {
|
|
const conn = deviceConnections.get(device.id);
|
|
|
|
// #146: a device with a live, still-connected socket is UP, even if its last
|
|
// heartbeat event is stuck behind a lagged event loop. Marking it offline on a
|
|
// stale in-memory lastHeartbeat was the second false-offline cause (the screen
|
|
// is online and playing, the CMS says offline). The socket still being in the
|
|
// /device namespace is the authoritative liveness signal — trust it over the
|
|
// (possibly queued) heartbeat clock. If the socket is genuinely gone, conn is
|
|
// either absent or points at a socket no longer in the namespace, and we fall
|
|
// through to the timeout below.
|
|
if (conn && deviceNs.sockets.has(conn.socketId)) continue;
|
|
|
|
const lastBeat = conn ? conn.lastHeartbeat : (device.last_heartbeat ? device.last_heartbeat * 1000 : 0);
|
|
|
|
if (now - lastBeat > config.heartbeatTimeout) {
|
|
db.prepare("UPDATE devices SET status = 'offline', updated_at = strftime('%s','now') WHERE id = ?")
|
|
.run(device.id);
|
|
deviceConnections.delete(device.id);
|
|
|
|
// Notify dashboard (workspace-scoped via the device's room).
|
|
emitToWorkspace(dashboardNs, deviceRoom(device.id), 'dashboard:device-status', {
|
|
device_id: device.id,
|
|
status: 'offline',
|
|
telemetry: null
|
|
});
|
|
|
|
console.log(`Device ${device.id} marked offline (heartbeat timeout)`);
|
|
// #146: batch through the coalescing writer (was an immediate INSERT here).
|
|
statusLogWriter.record(device.id, 'offline_timeout');
|
|
}
|
|
}
|
|
|
|
// #146: all table-growth maintenance runs OFF the interval body — async, chunked,
|
|
// band-gated, re-entrancy-guarded — so a sweep can never block the loop or stack.
|
|
// The offline-marking above stays synchronous (it's the core heartbeat function).
|
|
runMaintenance();
|
|
|
|
}, config.heartbeatInterval);
|
|
}
|
|
|
|
// #146: batched play-log prune (idx_play_logs_time), chunked so a 90-day backlog
|
|
// trims across many bounded DELETEs instead of one large statement.
|
|
const _delPlayLogs = db.prepare('DELETE FROM play_logs WHERE rowid IN (SELECT rowid FROM play_logs WHERE started_at < ? LIMIT ?)');
|
|
async function prunePlayLogs() {
|
|
const cutoff = Math.floor(Date.now() / 1000) - (90 * 86400);
|
|
return (await chunkedDelete((lim) => _delPlayLogs.run(cutoff, lim).changes, { batch: config.statusLogPruneBatch })).deleted;
|
|
}
|
|
|
|
// #146 interval maintenance — band-gated (skip while loaded; runs next tick) and
|
|
// re-entrancy-guarded (a long run never stacks with the next interval). Never throws
|
|
// into the interval. NOT for startup (see the un-gated startup prune above).
|
|
let _maintRunning = false;
|
|
async function runMaintenance() {
|
|
if (_maintRunning) return;
|
|
if (currentBand() !== 'normal') return;
|
|
_maintRunning = true;
|
|
try {
|
|
await pruneProvisioningDevices();
|
|
await prunePlayLogs();
|
|
await pruneStatusLog({ bandGate: true }); // per-device chunked; own re-entrancy
|
|
// Expiry sweeps on small tables — single cheap statements, bounded by table size.
|
|
db.prepare("DELETE FROM team_invites WHERE expires_at < strftime('%s','now')").run();
|
|
db.prepare("DELETE FROM workspace_invites WHERE expires_at < strftime('%s','now')").run();
|
|
} catch (_) { /* maintenance must never crash the interval */ } finally { _maintRunning = false; }
|
|
}
|
|
|
|
function registerConnection(deviceId, socketId) {
|
|
deviceConnections.set(deviceId, { socketId, lastHeartbeat: Date.now() });
|
|
}
|
|
|
|
function updateHeartbeat(deviceId) {
|
|
const conn = deviceConnections.get(deviceId);
|
|
if (conn) conn.lastHeartbeat = Date.now();
|
|
}
|
|
|
|
function removeConnection(deviceId) {
|
|
deviceConnections.delete(deviceId);
|
|
}
|
|
|
|
function getConnection(deviceId) {
|
|
return deviceConnections.get(deviceId);
|
|
}
|
|
|
|
function getAllConnections() {
|
|
return deviceConnections;
|
|
}
|
|
|
|
// #142: sweep unclaimed provisioning devices older than 24h (imported devices keep a
|
|
// user_id and are preserved). #146: now async + CHUNKED (rides idx_devices_provisioning)
|
|
// so a provisioning-junk flood can't delete-cascade a huge batch in one synchronous
|
|
// statement. Returns rows deleted. NOTE: async now — callers must await.
|
|
const _delProvisioning = db.prepare(`
|
|
DELETE FROM devices WHERE rowid IN (
|
|
SELECT rowid FROM devices
|
|
WHERE status = 'provisioning' AND user_id IS NULL AND created_at < ?
|
|
LIMIT ?
|
|
)
|
|
`);
|
|
async function pruneProvisioningDevices() {
|
|
const cutoff = Math.floor(Date.now() / 1000) - (24 * 3600);
|
|
return (await chunkedDelete((lim) => _delProvisioning.run(cutoff, lim).changes, { batch: config.statusLogPruneBatch })).deleted;
|
|
}
|
|
|
|
module.exports = {
|
|
startHeartbeatChecker,
|
|
registerConnection,
|
|
updateHeartbeat,
|
|
removeConnection,
|
|
getConnection,
|
|
getAllConnections,
|
|
pruneProvisioningDevices
|
|
};
|