mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-10-03 05:59:43 +02:00
fix: avoid event-loop stalls from synchronous tmux/ps calls (#100)
The stats collector (~2s) and mouse-mode sync (5s) ran execSync (pgrep/ps/ list-panes, 5s timeout each) per session on the server's single thread, blocking the event loop. With several sessions or a momentarily slow tmux this froze port 3000 for seconds-to-tens-of-seconds while the process stayed alive and other ports were unaffected — self-healing, so it never restarted and the 60s loopback healthcheck missed it. Convert these hot-path calls to execAsync. Also add an always-on event-loop lag monitor (utils/event-loop-monitor.ts) that logs stalls >=1s to the web log, so this otherwise-invisible class of incident leaves a quantified, timestamped trace. Co-authored-by: Teigen <teigen@TeigendeMac-mini.local>
This commit is contained in:
@@ -0,0 +1,54 @@
|
||||
/**
|
||||
* @fileoverview Event-loop lag monitor.
|
||||
*
|
||||
* Node is single-threaded: any synchronous work (e.g. a blocking `execSync`)
|
||||
* freezes the whole event loop, so the HTTP server stops answering on its port
|
||||
* while the process stays alive and other ports are unaffected. Such stalls
|
||||
* self-heal and never restart the process, so a periodic loopback healthcheck
|
||||
* misses them entirely — they leave no trace.
|
||||
*
|
||||
* This monitor samples how late a fixed-interval timer actually fires versus when
|
||||
* it was scheduled; the excess is time the loop was blocked. When that exceeds a
|
||||
* threshold it logs the measured stall, turning otherwise-invisible "port briefly
|
||||
* unreachable" incidents into a timestamped, quantified log line.
|
||||
*
|
||||
* @module utils/event-loop-monitor
|
||||
*/
|
||||
|
||||
export interface EventLoopMonitorHandle {
|
||||
stop(): void;
|
||||
}
|
||||
|
||||
/**
|
||||
* Start sampling event-loop lag.
|
||||
*
|
||||
* @param sampleMs How often to sample (and the baseline interval lag is measured against).
|
||||
* @param thresholdMs Only stalls at or above this many ms are logged (noise floor).
|
||||
* @param log Sink for stall reports; defaults to console.warn (lands in the web log).
|
||||
*/
|
||||
export function startEventLoopMonitor(
|
||||
sampleMs = 1000,
|
||||
thresholdMs = 1000,
|
||||
log: (msg: string) => void = (m) => console.warn(m)
|
||||
): EventLoopMonitorHandle {
|
||||
let last = performance.now();
|
||||
|
||||
const timer = setInterval(() => {
|
||||
const now = performance.now();
|
||||
// Lag = elapsed beyond the scheduled interval = time the loop was blocked.
|
||||
const lag = Math.round(now - last - sampleMs);
|
||||
if (lag >= thresholdMs) {
|
||||
log(`[EventLoopLag] event loop blocked ~${lag}ms (at ${new Date().toISOString()})`);
|
||||
}
|
||||
last = now;
|
||||
}, sampleMs);
|
||||
|
||||
// Never keep the process alive solely for this monitor.
|
||||
timer.unref?.();
|
||||
|
||||
return {
|
||||
stop() {
|
||||
clearInterval(timer);
|
||||
},
|
||||
};
|
||||
}
|
||||
@@ -9,6 +9,8 @@
|
||||
export { BufferAccumulator } from './buffer-accumulator.js';
|
||||
export { CleanupManager } from './cleanup-manager.js';
|
||||
export { Debouncer, KeyedDebouncer } from './debouncer.js';
|
||||
export { startEventLoopMonitor } from './event-loop-monitor.js';
|
||||
export type { EventLoopMonitorHandle } from './event-loop-monitor.js';
|
||||
export { StaleExpirationMap } from './stale-expiration-map.js';
|
||||
export {
|
||||
ANSI_ESCAPE_PATTERN_FULL,
|
||||
|
||||
Reference in New Issue
Block a user