From 83cb0b46b15f25569f74cd300b1ac0a3c6d0276e Mon Sep 17 00:00:00 2001 From: Codeman maintainer Date: Wed, 23 Sep 2026 20:26:48 +0200 Subject: [PATCH] fix(self-update): a hung graceful shutdown no longer leaves a LaunchDaemon install down On a KeepAlive LaunchDaemon (headless macOS) the updater restarts by sending the server SIGTERM and letting launchd respawn it. launchd only respawns once the process EXITS, and nothing escalates a stuck stop (systemd would SIGKILL after TimeoutStopSec). Observed after an update to 1.32.1: the server closed port 3000, server.stop() never resolved, the process stayed alive and the service stayed down until it was killed by hand. - cli.ts: the signal handler arms an unref'd 10s timer that force-exits if server.stop() hangs. - self-update.sh (launchd-daemon): wait up to 30s for the server pid to exit, then SIGKILL it. tmux sessions live outside the server and survive. Co-Authored-By: Claude Opus 5.5 (1M context) --- scripts/self-update.sh | 12 +++++++++++- src/cli.ts | 10 ++++++++++ 2 files changed, 21 insertions(+), 1 deletion(-) diff --git a/scripts/self-update.sh b/scripts/self-update.sh index e89b7e59..6ca3dd7b 100755 --- a/scripts/self-update.sh +++ b/scripts/self-update.sh @@ -285,7 +285,17 @@ case "$SUPERVISOR" in # domain needs root, but we don't need it — kill the server and launchd # respawns it on the new dist/ within ThrottleInterval seconds. if [[ -n "$SERVER_PID" ]] && kill "$SERVER_PID" 2>/dev/null; then - : # respawn is launchd's job from here + # Respawn is launchd's job, but only once the old process EXITS. A graceful + # shutdown that hangs leaves the port closed and the service down, so + # escalate to SIGKILL (tmux sessions live outside the server and survive). + for _ in $(seq 1 30); do + kill -0 "$SERVER_PID" 2>/dev/null || break + sleep 1 + done + if kill -0 "$SERVER_PID" 2>/dev/null; then + echo "[self-update] server pid $SERVER_PID still alive 30s after SIGTERM, sending SIGKILL" + kill -9 "$SERVER_PID" 2>/dev/null || true + fi else MANUAL_CMD="sudo launchctl kickstart -k system/com.codeman.web" write_status "completed-needs-manual-restart" "Update staged — restart Codeman to apply v$TO_VERSION." diff --git a/src/cli.ts b/src/cli.ts index 2da7fac7..1b60246c 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -129,6 +129,8 @@ program /** Same registry the server resolves case names through (mirrors `case-routes.ts`). */ const LINKED_CASES_FILE = dataPath('linked-cases.json'); +/** Graceful shutdown budget before the process force-exits (see the SIGTERM handler). */ +const SHUTDOWN_FORCE_EXIT_MS = 10_000; /** * Case name to directory, checking `linked-cases.json` FIRST and falling back to the @@ -1002,6 +1004,14 @@ webCmd.action(async (options) => { if (shuttingDown) return; shuttingDown = true; console.log(palette.warn(`\n${signal} received, shutting down gracefully...`)); + // A hung stop() must not keep the process alive: the listener is already + // closed by then, and a KeepAlive LaunchDaemon only respawns the server once + // it EXITS (systemd would SIGKILL after TimeoutStopSec; launchd does not). + // Seen after a self-update on macOS: port closed, process alive, service down. + setTimeout(() => { + console.error(palette.err(`Shutdown did not finish in ${SHUTDOWN_FORCE_EXIT_MS / 1000}s, forcing exit`)); + process.exit(1); + }, SHUTDOWN_FORCE_EXIT_MS).unref(); try { await server.stop(); } catch (err) {