mirror of
https://github.com/screentinker/screentinker.git
synced 2026-08-14 06:16:20 -06:00
Found in the alpha load test: client-chosen pairing codes collide by birthday paradox, the provisioning INSERT hit UNIQUE(devices.pairing_code), the SqliteError threw out of the (synchronous) socket handler -> uncaughtException -> logFatalAndExit -> the WHOLE server exited and every device dropped. The colliding flood crash-LOOPED the container (2 restarts). Two layers, same "one device can't take down the fleet" theme as #142/#143/#144: 1. Narrow (deviceSocket.js): wrap the device:register provisioning INSERT in try/catch — a UNIQUE pairing_code collision (or ANY db error) rejects THAT registration (device:auth-error -> client retries) instead of throwing. currentDeviceId/authenticated now set only AFTER the row exists (no half-auth socket on failure). 2. Broader (lib/safe-socket.js): protectSocket() overrides socket.on per connection so any handler throw is caught, logged (event + id + stack), the socket told, and DISCONNECTED — per-CONNECTION fail-fast, not whole-PROCESS. We don't keep serving a connection from possibly-half-mutated state (honors the existing fail-fast intent), we just contain it to "one device reconnects" (a non-event after beta5). Wired into both the /device and /dashboard connection handlers; auto-covers future handlers. Audited first: no handler throws as control flow, so blanket-wrapping is safe. Tests (mutation-verified, fail without their fix): - register-insert-crash.test.js: a pairing_code collision AND a general bind error each reject-one-device with no uncaughtException; server keeps serving. - socket-handler-isolation.test.js: a throwing handler disconnects only that socket; the server + other sockets stay alive. Full suite 243/243. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
133 lines
6 KiB
JavaScript
133 lines
6 KiB
JavaScript
const heartbeat = require('../services/heartbeat');
|
|
const { verifyToken } = require('../middleware/auth');
|
|
const { db } = require('../db/database');
|
|
const { accessContext, accessibleWorkspaceIds } = require('../lib/tenancy');
|
|
const { workspaceRoom } = require('../lib/socket-rooms');
|
|
const { protectSocket } = require('../lib/safe-socket');
|
|
|
|
// Phase 2.3: workspace-scoped socket rooms + per-command permission gates.
|
|
// Replaces the previous flat dashboardNs.emit broadcast (which leaked every
|
|
// device's status/screenshot/playback events to every connected dashboard)
|
|
// and the legacy admin/superadmin role bypass (dead code post-Phase-1
|
|
// rename - admin -> user, superadmin -> platform_admin).
|
|
//
|
|
// On connect: enumerate the user's accessible workspace_ids and socket.join
|
|
// a room per workspace. Outbound broadcasts route via dashboardNs.to(room).
|
|
// Inbound commands check permission against the target device's workspace.
|
|
|
|
// Permission gate for inbound socket commands. Read tier = workspace_viewer+;
|
|
// write tier = workspace_editor+. Platform_admin and org_owner/admin always
|
|
// pass via actingAs.
|
|
function canActOnDevice(socket, deviceId, tier /* 'read' | 'write' */) {
|
|
const device = db.prepare('SELECT workspace_id FROM devices WHERE id = ?').get(deviceId);
|
|
if (!device || !device.workspace_id) return false;
|
|
const ws = db.prepare('SELECT * FROM workspaces WHERE id = ?').get(device.workspace_id);
|
|
if (!ws) return false;
|
|
const ctx = accessContext(socket.userId, socket.userRole, ws);
|
|
if (!ctx) return false;
|
|
if (ctx.actingAs) return true; // platform_admin or org admin
|
|
if (tier === 'read') return !!ctx.workspaceRole; // viewer/editor/admin all OK
|
|
// write tier: workspace_editor or workspace_admin
|
|
return ctx.workspaceRole === 'workspace_editor' || ctx.workspaceRole === 'workspace_admin';
|
|
}
|
|
|
|
module.exports = function setupDashboardSocket(io) {
|
|
const dashboardNs = io.of('/dashboard');
|
|
const deviceNs = io.of('/device');
|
|
|
|
dashboardNs.use((socket, next) => {
|
|
const token = socket.handshake.auth?.token;
|
|
if (!token) return next(new Error('Authentication required'));
|
|
try {
|
|
const decoded = verifyToken(token);
|
|
socket.userId = decoded.id;
|
|
socket.userRole = decoded.role;
|
|
next();
|
|
} catch {
|
|
next(new Error('Invalid token'));
|
|
}
|
|
});
|
|
|
|
dashboardNs.on('connection', (socket) => {
|
|
// #146: same per-connection fail-fast as the device namespace — a throwing
|
|
// dashboard handler disconnects only that client, never crashes the server.
|
|
protectSocket(socket, () => socket.userId);
|
|
// Note on workspace-switch lifecycle: the switcher (Phase 3 MVP) calls
|
|
// window.location.reload() after switching, which forces a new socket
|
|
// connection with fresh JWT claims. So workspace memberships are
|
|
// re-evaluated at connect time and we don't need to re-evaluate per-emit.
|
|
const wsIds = accessibleWorkspaceIds(socket.userId, socket.userRole);
|
|
for (const wsId of wsIds) socket.join(workspaceRoom(wsId));
|
|
console.log(`Dashboard client connected: ${socket.id} (user: ${socket.userId}, rooms: ${wsIds.length})`);
|
|
|
|
socket.on('dashboard:request-screenshot', (data) => {
|
|
const { device_id } = data;
|
|
if (!canActOnDevice(socket, device_id, 'read')) return;
|
|
const conn = heartbeat.getConnection(device_id);
|
|
if (conn) deviceNs.to(device_id).emit('device:screenshot-request', {});
|
|
});
|
|
|
|
socket.on('dashboard:remote-touch', (data) => {
|
|
const { device_id, x, y, action } = data;
|
|
if (!canActOnDevice(socket, device_id, 'write')) return;
|
|
deviceNs.to(device_id).emit('device:remote-touch', { x, y, action });
|
|
});
|
|
|
|
socket.on('dashboard:remote-key', (data) => {
|
|
const { device_id, keycode } = data;
|
|
if (!canActOnDevice(socket, device_id, 'write')) return;
|
|
console.log(`Remote key: ${keycode} -> ${device_id}`);
|
|
deviceNs.to(device_id).emit('device:remote-key', { keycode });
|
|
});
|
|
|
|
socket.on('dashboard:remote-start', (data) => {
|
|
const { device_id } = data;
|
|
if (!canActOnDevice(socket, device_id, 'write')) return;
|
|
const room = deviceNs.adapter.rooms.get(device_id);
|
|
console.log(`Remote start for ${device_id}, room has ${room?.size || 0} socket(s)`);
|
|
deviceNs.to(device_id).emit('device:remote-start', {});
|
|
console.log(`Remote session started for device ${device_id}`);
|
|
});
|
|
|
|
socket.on('dashboard:remote-stop', (data) => {
|
|
const { device_id } = data;
|
|
if (!canActOnDevice(socket, device_id, 'write')) return;
|
|
deviceNs.to(device_id).emit('device:remote-stop', {});
|
|
console.log(`Remote session stopped for device ${device_id}`);
|
|
});
|
|
|
|
socket.on('dashboard:device-command', (data, ack) => {
|
|
const { device_id, type, payload } = data;
|
|
if (!canActOnDevice(socket, device_id, 'write')) {
|
|
if (typeof ack === 'function') ack({ delivered: false, reason: 'forbidden' });
|
|
return;
|
|
}
|
|
const room = deviceNs.adapter.rooms.get(device_id);
|
|
if (room && room.size > 0) {
|
|
deviceNs.to(device_id).emit('device:command', { type, payload });
|
|
console.log(`Command delivered to device ${device_id}: ${type}`);
|
|
if (typeof ack === 'function') ack({ delivered: true });
|
|
return;
|
|
}
|
|
// Device offline at emit time. Try to queue (lazy require so reverting
|
|
// the queue commit doesn't break this commit - MODULE_NOT_FOUND on the
|
|
// first try gets cached by Node's module loader, giving consistent
|
|
// queued=false behavior on every subsequent call).
|
|
let queued = false;
|
|
try {
|
|
const queue = require('../lib/command-queue');
|
|
queued = queue.queueCommand(device_id, type, payload);
|
|
} catch (e) { /* command-queue module absent; fall through to lost */ }
|
|
console.log(`Command for offline device ${device_id}: ${type} (queued=${queued})`);
|
|
if (typeof ack === 'function') ack({ delivered: false, queued, reason: 'offline' });
|
|
});
|
|
|
|
socket.on('disconnect', () => {
|
|
console.log(`Dashboard client disconnected: ${socket.id}`);
|
|
});
|
|
});
|
|
|
|
return dashboardNs;
|
|
};
|
|
|