screentinker/server/services/heartbeat.js
ScreenTinker 9418582de5 feat(#146): always-on devices_connected + admin-toggleable /api/status debug block
1. devices_connected (always on, never gated): a top-level /api/status field next to
   loop_lag = LIVE WS socket count from the heartbeat connection map (getConnectedCount),
   NOT devices.status='online' (which lags by the offline-timeout). The single
   most-glanced operational number, so it can't disappear when debug is off. Also dropped
   4 dead per-poll COUNT(*) queries the route computed but never returned.

2. debug block behind an admin flag: new minimal app_settings KV table (none existed;
   ai_settings is per-workspace, white_labels is branding) + lib/app-settings.js (cached,
   refresh-on-write so status polls read a cached boolean, not a DB row).
   routes/status.js includes `debug` ONLY when status_debug_enabled is on (persisted value
   overrides the STATUS_DEBUG_ENABLED env default); when off the key is omitted entirely.

3. Admin toggle: GET/PUT /api/admin/status-debug (requirePlatformAdmin, mirrors the
   branding endpoints) + a checkbox in the Admin tab "Status endpoint" section
   (mirrors the branding checkbox). Takes effect on the next poll, no restart.

Tests: devices_connected always present+numeric and rises with a live socket (booted +
socket.io-client); debug present by default, admin flips OFF -> key omitted (loop_lag +
devices_connected remain) -> ON again, no restart; non-admin 403, anon 401; unit coverage
for getConnectedCount + app-settings default/override. Suite 289/289.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-01 18:45:40 -05:00

150 lines
6.3 KiB
JavaScript

const { db, pruneStatusLog } = require('../db/database');
const config = require('../config');
const { deviceRoom, emitToWorkspace } = require('../lib/socket-rooms');
const statusLogWriter = require('../lib/status-log-writer');
const { chunkedDelete, currentBand } = require('../lib/chunked-prune'); // #146 non-blocking sweeps
// Track connected device sockets: deviceId -> { socketId, lastHeartbeat }
const deviceConnections = new Map();
function startHeartbeatChecker(io) {
// #146: startup sweep is chunked + async + fire-and-forget + NOT band-gated, so a
// bloated device_status_log self-heals on next deploy WITHOUT freezing boot (the old
// whole-table sort froze boot 40-48s -> healthcheck fail -> restart loop). It
// trickles in bounded batches while the server comes up and serves.
pruneStatusLog({ bandGate: false }).catch(() => {});
// #146: start the batched device_status_log flush loop.
statusLogWriter.start();
const deviceNs = io.of('/device');
setInterval(() => {
const now = Date.now();
const dashboardNs = io.of('/dashboard');
// Check database for devices that should be offline
const onlineDevices = db.prepare("SELECT id, last_heartbeat FROM devices WHERE status = 'online'").all();
for (const device of onlineDevices) {
const conn = deviceConnections.get(device.id);
// #146: a device with a live, still-connected socket is UP, even if its last
// heartbeat event is stuck behind a lagged event loop. Marking it offline on a
// stale in-memory lastHeartbeat was the second false-offline cause (the screen
// is online and playing, the CMS says offline). The socket still being in the
// /device namespace is the authoritative liveness signal — trust it over the
// (possibly queued) heartbeat clock. If the socket is genuinely gone, conn is
// either absent or points at a socket no longer in the namespace, and we fall
// through to the timeout below.
if (conn && deviceNs.sockets.has(conn.socketId)) continue;
const lastBeat = conn ? conn.lastHeartbeat : (device.last_heartbeat ? device.last_heartbeat * 1000 : 0);
if (now - lastBeat > config.heartbeatTimeout) {
db.prepare("UPDATE devices SET status = 'offline', updated_at = strftime('%s','now') WHERE id = ?")
.run(device.id);
deviceConnections.delete(device.id);
// Notify dashboard (workspace-scoped via the device's room).
emitToWorkspace(dashboardNs, deviceRoom(device.id), 'dashboard:device-status', {
device_id: device.id,
status: 'offline',
telemetry: null
});
console.log(`Device ${device.id} marked offline (heartbeat timeout)`);
// #146: batch through the coalescing writer (was an immediate INSERT here).
statusLogWriter.record(device.id, 'offline_timeout');
}
}
// #146: all table-growth maintenance runs OFF the interval body — async, chunked,
// band-gated, re-entrancy-guarded — so a sweep can never block the loop or stack.
// The offline-marking above stays synchronous (it's the core heartbeat function).
runMaintenance();
}, config.heartbeatInterval);
}
// #146: batched play-log prune (idx_play_logs_time), chunked so a 90-day backlog
// trims across many bounded DELETEs instead of one large statement.
const _delPlayLogs = db.prepare('DELETE FROM play_logs WHERE rowid IN (SELECT rowid FROM play_logs WHERE started_at < ? LIMIT ?)');
async function prunePlayLogs() {
const cutoff = Math.floor(Date.now() / 1000) - (90 * 86400);
return (await chunkedDelete((lim) => _delPlayLogs.run(cutoff, lim).changes, { batch: config.statusLogPruneBatch })).deleted;
}
// #146 interval maintenance — band-gated (skip while loaded; runs next tick) and
// re-entrancy-guarded (a long run never stacks with the next interval). Never throws
// into the interval. NOT for startup (see the un-gated startup prune above).
let _maintRunning = false;
async function runMaintenance() {
if (_maintRunning) return;
if (config.maintenanceBandGateEnabled && currentBand() !== 'normal') return; // #146 P1.3 kill switch
_maintRunning = true;
try {
await pruneProvisioningDevices();
await prunePlayLogs();
await pruneStatusLog({ bandGate: true }); // per-device chunked; own re-entrancy
// Expiry sweeps on small tables — single cheap statements, bounded by table size.
db.prepare("DELETE FROM team_invites WHERE expires_at < strftime('%s','now')").run();
db.prepare("DELETE FROM workspace_invites WHERE expires_at < strftime('%s','now')").run();
} catch (_) { /* maintenance must never crash the interval */ } finally { _maintRunning = false; }
}
function registerConnection(deviceId, socketId) {
deviceConnections.set(deviceId, { socketId, lastHeartbeat: Date.now() });
}
function updateHeartbeat(deviceId) {
const conn = deviceConnections.get(deviceId);
if (conn) conn.lastHeartbeat = Date.now();
}
function removeConnection(deviceId) {
deviceConnections.delete(deviceId);
}
function getConnection(deviceId) {
return deviceConnections.get(deviceId);
}
function getAllConnections() {
return deviceConnections;
}
// #146: LIVE connected-device count — the set with a live socket THIS INSTANT. Cheap
// in-memory read. Distinct from devices.status='online' (persisted, lags by the
// offline-timeout). Surfaced as /api/status.devices_connected.
function getConnectedCount() {
return deviceConnections.size;
}
// #142: sweep unclaimed provisioning devices older than 24h (imported devices keep a
// user_id and are preserved). #146: now async + CHUNKED (rides idx_devices_provisioning)
// so a provisioning-junk flood can't delete-cascade a huge batch in one synchronous
// statement. Returns rows deleted. NOTE: async now — callers must await.
const _delProvisioning = db.prepare(`
DELETE FROM devices WHERE rowid IN (
SELECT rowid FROM devices
WHERE status = 'provisioning' AND user_id IS NULL AND created_at < ?
LIMIT ?
)
`);
async function pruneProvisioningDevices() {
const cutoff = Math.floor(Date.now() / 1000) - (24 * 3600);
return (await chunkedDelete((lim) => _delProvisioning.run(cutoff, lim).changes, { batch: config.statusLogPruneBatch })).deleted;
}
module.exports = {
startHeartbeatChecker,
registerConnection,
updateHeartbeat,
removeConnection,
getConnection,
getAllConnections,
getConnectedCount,
pruneProvisioningDevices
};