@@ -3535,6 +3553,7 @@
+
diff --git a/src/web/public/reboot-restore-ui.js b/src/web/public/reboot-restore-ui.js
new file mode 100644
index 00000000..0869eebe
--- /dev/null
+++ b/src/web/public/reboot-restore-ui.js
@@ -0,0 +1,95 @@
+/**
+ * @fileoverview Reboot-restore banner: offer back the sessions a host reboot destroyed.
+ *
+ * A host reboot takes the tmux server down with it, so every session's pane dies
+ * and the board comes up empty. The server works out what was running from the
+ * records it still holds at boot, and this banner asks the user whether to
+ * rebuild them. Nothing is created until they click, because the server's
+ * reboot guess is a heuristic and a wrong automatic restore would spawn CLI
+ * processes nobody asked for.
+ *
+ * Seeded once from `GET /api/reboot-restore` on init. Restore posts to
+ * `POST /api/reboot-restore/restore`, Dismiss posts to
+ * `POST /api/reboot-restore/dismiss`, and either way the banner goes away. The
+ * restored sessions arrive as ordinary `session:created` events, so no extra
+ * rendering is needed here.
+ *
+ * The banner says that terminal history did not survive, because a restored
+ * session is a new pane: the conversation continues and the scrollback does not.
+ * Saying so is what keeps an empty pane from reading as a broken restore.
+ * Backend: src/web/reboot-restore-registry.ts, src/web/routes/reboot-restore-routes.ts.
+ *
+ * @mixin Extends CodemanApp.prototype via Object.assign
+ * @dependency app.js (CodemanApp class, showToast)
+ * @dependency api-client.js at runtime (this._apiJson / this._apiPost)
+ * @loadorder 11.7 of 17, after approvals-ui.js
+ */
+
+Object.assign(CodemanApp.prototype, {
+ /** Ask the server whether a reboot left anything on offer, and show the banner if so. */
+ async initRebootRestoreBanner() {
+ const data = await this._apiJson('/api/reboot-restore');
+ const sessions = data?.sessions ?? [];
+ if (sessions.length === 0) return;
+ this._rebootRestoreSessions = sessions;
+ this.renderRebootRestoreBanner();
+ },
+
+ renderRebootRestoreBanner() {
+ const banner = this.$('rebootRestoreBanner');
+ if (!banner) return;
+ const sessions = this._rebootRestoreSessions ?? [];
+ if (sessions.length === 0) {
+ banner.hidden = true;
+ return;
+ }
+ const count = sessions.length;
+ const text = this.$('rebootRestoreBannerText');
+ if (text) {
+ const noun = count === 1 ? 'session' : 'sessions';
+ text.textContent = `Restore ${count} ${noun} from before the reboot`;
+ }
+ const detail = this.$('rebootRestoreBannerDetail');
+ if (detail) {
+ // Names, so the user can tell what they are about to relaunch.
+ const names = sessions
+ .map((s) => s.name || s.workingDir?.split('/').pop() || s.id.slice(0, 8))
+ .slice(0, 4)
+ .join(', ');
+ detail.textContent = count > 4 ? `${names}, …` : names;
+ detail.title = sessions.map((s) => `${s.name || s.id}\n${s.workingDir}`).join('\n\n');
+ }
+ banner.hidden = false;
+ },
+
+ /** Rebuild everything on offer. The panes are new, so scrollback does not come back. */
+ async restoreRebootSessions() {
+ const button = this.$('rebootRestoreBannerAccept');
+ if (button) button.disabled = true;
+ const res = await this._apiPost('/api/reboot-restore/restore', {});
+ const body = res && res.ok ? await res.json().catch(() => null) : null;
+ if (!body) {
+ if (button) button.disabled = false;
+ this.showToast?.('Could not restore the sessions', 'error');
+ return;
+ }
+ const restored = body.restored?.length ?? 0;
+ const skipped = body.skipped?.length ?? 0;
+ this._rebootRestoreSessions = [];
+ this.renderRebootRestoreBanner();
+ if (restored > 0) {
+ const noun = restored === 1 ? 'conversation' : 'conversations';
+ this.showToast?.(`Restored ${restored} ${noun}. Terminal history did not survive the reboot.`, 'success');
+ }
+ if (skipped > 0) {
+ this.showToast?.(`${skipped} could not be restored (workspace gone, or already open)`, 'warning');
+ }
+ },
+
+ /** Drop the offer. The Resume list still reaches every one of these conversations. */
+ async dismissRebootRestore() {
+ this._rebootRestoreSessions = [];
+ this.renderRebootRestoreBanner();
+ await this._apiPost('/api/reboot-restore/dismiss', {});
+ },
+});
diff --git a/src/web/public/styles.css b/src/web/public/styles.css
index 398eb6ce..b5e4873b 100644
--- a/src/web/public/styles.css
+++ b/src/web/public/styles.css
@@ -15243,6 +15243,81 @@ html[data-skin="daylight-blue"] .welcome-btn-tunnel.active:hover {
skin, including the light ones. Visibility is driven by the `hidden`
attribute, so the display rules need !important to lose to it. */
+/* Reboot-restore offer. Amber rather than red: nothing is wrong, the board is
+ asking a question, and the user can ignore it. See reboot-restore-ui.js. */
+.reboot-restore-banner {
+ display: flex;
+ align-items: center;
+ gap: 0.6rem;
+ padding: 0.45rem 1rem;
+ background: linear-gradient(90deg, #b45309, #92400e);
+ border-bottom: 1px solid rgba(0, 0, 0, 0.35);
+ color: #fff;
+ font-size: 0.78rem;
+ font-weight: 600;
+ letter-spacing: 0.01em;
+ flex-shrink: 0;
+ z-index: 1250;
+}
+
+.reboot-restore-banner[hidden] {
+ display: none !important;
+}
+
+.reboot-restore-banner-icon {
+ flex-shrink: 0;
+ font-size: 0.95rem;
+ line-height: 1;
+}
+
+.reboot-restore-banner-text {
+ white-space: nowrap;
+}
+
+.reboot-restore-banner-detail {
+ color: rgba(255, 255, 255, 0.8);
+ font-weight: 500;
+ overflow: hidden;
+ text-overflow: ellipsis;
+ white-space: nowrap;
+}
+
+.reboot-restore-banner-note {
+ color: rgba(255, 255, 255, 0.75);
+ font-weight: 500;
+ white-space: nowrap;
+ margin-left: auto;
+}
+
+.reboot-restore-banner-accept,
+.reboot-restore-banner-dismiss {
+ flex-shrink: 0;
+ padding: 0.2rem 0.6rem;
+ border-radius: 5px;
+ border: 1px solid rgba(255, 255, 255, 0.55);
+ background: rgba(255, 255, 255, 0.12);
+ color: #fff;
+ font-size: 0.72rem;
+ font-weight: 600;
+ cursor: pointer;
+}
+
+.reboot-restore-banner-accept:hover,
+.reboot-restore-banner-dismiss:hover {
+ background: rgba(255, 255, 255, 0.24);
+}
+
+.reboot-restore-banner-accept:disabled {
+ opacity: 0.6;
+ cursor: default;
+}
+
+.reboot-restore-banner-dismiss {
+ border-color: rgba(255, 255, 255, 0.3);
+ background: transparent;
+ font-weight: 500;
+}
+
.offline-banner {
display: flex;
align-items: center;
diff --git a/src/web/reboot-restore-registry.ts b/src/web/reboot-restore-registry.ts
new file mode 100644
index 00000000..e265fa62
--- /dev/null
+++ b/src/web/reboot-restore-registry.ts
@@ -0,0 +1,142 @@
+/**
+ * @fileoverview The pending restore plan: what a host reboot destroyed, waiting on a click.
+ *
+ * The boot pass builds this plan inside `restoreMuxSessions()`, in the window
+ * where reconciliation has reported the dead sessions and `cleanupStaleSessions()`
+ * has not pruned their records yet. The board then offers "restore N sessions
+ * from before the reboot", and `web/routes/reboot-restore-routes` spends the plan
+ * when the user clicks.
+ *
+ * Invariants:
+ * - Entries are in-memory only. A server restart drops the plan, and nothing
+ * re-builds it, because the records it was built from are pruned by then.
+ * That costs the convenience this feature adds and never the conversation:
+ * the conversation IS the transcript under `~/.claude/projects`, which
+ * `services/unified-session-service.ts` reads for the Welcome screen's Resume
+ * list and the Session Manager, and `resumeHistorySession()` in
+ * `web/public/terminal-ui.js` resumes from a row there with no persisted
+ * session record involved. A dropped plan therefore returns the user to
+ * resuming by hand, one at a time, which is where they are without this
+ * feature. What the plan held that a transcript does not is the owner, the
+ * name, the env overrides, the effort and the lineage.
+ * - Module-level singleton in the style of `web/approval-inbox.ts`: no `Session`
+ * import and no IO, which keeps it unit-testable and cycle-free.
+ * - Spending is take-then-build: `take()` removes entries synchronously, before
+ * the route's first `await`, so a double-click or two devices cannot both
+ * reach the same entry and put two panes on one conversation.
+ * - One restore runs at a time. `beginSpending()` single-flights the route, so
+ * two concurrent clicks cannot interleave pane creation.
+ *
+ * @dependencies reboot-restore (RebootRestoreEntry)
+ * @consumedby web/server (plan build at boot), web/routes/reboot-restore-routes
+ *
+ * @module web/reboot-restore-registry
+ */
+
+import type { RebootRestoreEntry } from '../reboot-restore.js';
+
+/**
+ * A plan older than this is dropped on read. A machine that rebooted yesterday
+ * has moved on, and an offer nobody took by then is noise rather than a rescue.
+ */
+const PLAN_TTL_MS = 24 * 60 * 60 * 1000;
+
+export class RebootRestoreRegistry {
+ /** Keyed by session id, in the order the boot pass found them. */
+ private entries = new Map();
+ /** When the boot pass built the plan, in ms since the epoch. */
+ private builtAt = 0;
+ /** True while a restore route call is between its take and its last pane. */
+ private spending = false;
+
+ /** Replace the plan with what the boot pass found. An empty list clears it. */
+ set(entries: readonly RebootRestoreEntry[]): void {
+ this.entries = new Map(entries.map((entry) => [entry.sessionId, entry]));
+ this.builtAt = entries.length > 0 ? Date.now() : 0;
+ }
+
+ /**
+ * The entries a viewer may see, newest plan first-come order preserved.
+ *
+ * @param canAccess Ownership predicate, so a user sees their own entries and
+ * an admin sees all. Applied here rather than in the route so the count the
+ * banner shows and the entries a click spends come from one filter.
+ */
+ list(canAccess: (owner: string | undefined) => boolean): RebootRestoreEntry[] {
+ this.dropIfExpired();
+ return [...this.entries.values()].filter((entry) => canAccess(entry.owner));
+ }
+
+ /**
+ * Remove and return the entries a click is about to spend.
+ *
+ * Synchronous and total: an entry leaves the plan here, before any pane is
+ * created, so a second click finds nothing to spend. Entries a caller may not
+ * access are left in place, and unknown ids are ignored.
+ *
+ * @param sessionIds The ids to spend, or undefined for every visible entry.
+ */
+ take(canAccess: (owner: string | undefined) => boolean, sessionIds?: readonly string[]): RebootRestoreEntry[] {
+ this.dropIfExpired();
+ const wanted = sessionIds ? new Set(sessionIds) : undefined;
+ const taken: RebootRestoreEntry[] = [];
+ for (const entry of [...this.entries.values()]) {
+ if (wanted && !wanted.has(entry.sessionId)) continue;
+ if (!canAccess(entry.owner)) continue;
+ this.entries.delete(entry.sessionId);
+ taken.push(entry);
+ }
+ return taken;
+ }
+
+ /**
+ * Put entries back after a rebuild never got as far as creating a pane.
+ *
+ * Used for the click-time rejections, so a conversation the user resumed by
+ * hand meanwhile does not silently vanish from the banner while a workspace
+ * that came back stays offered.
+ */
+ restore(entries: readonly RebootRestoreEntry[]): void {
+ for (const entry of entries) this.entries.set(entry.sessionId, entry);
+ if (entries.length > 0 && this.builtAt === 0) this.builtAt = Date.now();
+ }
+
+ /** Drop the entries a viewer can see. Returns how many went. */
+ clear(canAccess: (owner: string | undefined) => boolean): number {
+ const removable = [...this.entries.values()].filter((entry) => canAccess(entry.owner));
+ for (const entry of removable) this.entries.delete(entry.sessionId);
+ if (this.entries.size === 0) this.builtAt = 0;
+ return removable.length;
+ }
+
+ /**
+ * Claim the right to run a restore, or report that one is already running.
+ * Callers that get `true` must call `endSpending()` in a `finally`.
+ */
+ beginSpending(): boolean {
+ if (this.spending) return false;
+ this.spending = true;
+ return true;
+ }
+
+ endSpending(): void {
+ this.spending = false;
+ }
+
+ /** Test hook: forget everything, including the single-flight claim. */
+ reset(): void {
+ this.entries.clear();
+ this.builtAt = 0;
+ this.spending = false;
+ }
+
+ private dropIfExpired(): void {
+ if (this.builtAt > 0 && Date.now() - this.builtAt > PLAN_TTL_MS) {
+ this.entries.clear();
+ this.builtAt = 0;
+ }
+ }
+}
+
+/** Process-wide singleton, mirroring `approvalInbox`. */
+export const rebootRestoreRegistry = new RebootRestoreRegistry();
diff --git a/src/web/routes/index.ts b/src/web/routes/index.ts
index c1f0a952..f11d6194 100644
--- a/src/web/routes/index.ts
+++ b/src/web/routes/index.ts
@@ -11,6 +11,7 @@ export { registerCronRoutes } from './cron-routes.js';
export { registerSystemRoutes } from './system-routes.js';
export { registerHookEventRoutes } from './hook-event-routes.js';
export { registerApprovalRoutes } from './approval-routes.js';
+export { registerRebootRestoreRoutes } from './reboot-restore-routes.js';
export { registerReadMyMindRoutes } from './readmymind-routes.js';
export { registerStatusTelemetryRoutes } from './status-telemetry-routes.js';
export { registerCaseRoutes } from './case-routes.js';
diff --git a/src/web/routes/reboot-restore-routes.ts b/src/web/routes/reboot-restore-routes.ts
new file mode 100644
index 00000000..580b17c6
--- /dev/null
+++ b/src/web/routes/reboot-restore-routes.ts
@@ -0,0 +1,194 @@
+/**
+ * @fileoverview Reboot-restore routes: offer back the sessions a host reboot destroyed.
+ *
+ * The boot pass leaves a plan in `web/reboot-restore-registry` when the machine
+ * plausibly rebooted. The board reads it, shows a banner, and the user decides:
+ * - `GET /api/reboot-restore`: what is on offer, ownership-scoped
+ * - `POST /api/reboot-restore/restore`: rebuild some or all of it
+ * - `POST /api/reboot-restore/dismiss`: drop the offer
+ *
+ * A click, not the heuristic, is what creates panes. The heuristic only decides
+ * whether the banner appears, so a wrong yes costs a line of text the user
+ * dismisses rather than N CLI processes nobody asked for.
+ *
+ * Rebuilding is take-then-build: entries leave the plan synchronously at the top
+ * of the route, before the first `await`, and the whole route is single-flighted,
+ * so a double-click or two devices cannot put two panes on one conversation.
+ * Three things are re-checked at click time rather than trusted from boot: the
+ * owner's privilege grant, the workspace still being on disk, and the
+ * conversation not already being live because the user resumed it by hand.
+ *
+ * A rebuilt session comes back attached, idle and disarmed. Respawn controllers
+ * and Ralph loops are deliberately not re-armed, and its terminal scrollback is
+ * gone, because the pane is new. The banner says so.
+ */
+
+import { FastifyInstance } from 'fastify';
+import { existsSync } from 'node:fs';
+import { ApiErrorCode, createErrorResponse, getErrorMessage } from '../../types.js';
+import { RebootRestoreRequestSchema } from '../schemas.js';
+import { parseBody, getAuthUser, canAccessOwned } from '../route-helpers.js';
+import { rebootRestoreRegistry } from '../reboot-restore-registry.js';
+import { rejectAlreadyLive, type RebootRestoreEntry, type RebootRestoreRejection } from '../../reboot-restore.js';
+import { clampEnvOverridesForOwner } from '../../session-env-clamp.js';
+import { Session } from '../../session.js';
+import { resolveClaudeModeForUsername } from '../../user-store.js';
+import { getCli } from '../../config/cli-registry/registry.js';
+import { applyWorkspaceHooks } from '../../hooks-config.js';
+import { getLifecycleLog } from '../../session-lifecycle-log.js';
+import { STATS_COLLECTION_INTERVAL_MS } from '../../config/server-timing.js';
+import { SseEvent } from '../sse-events.js';
+import type { SessionAttachmentHistoryItem } from '../../types.js';
+import type { SessionPort, EventPort, ConfigPort, InfraPort } from '../ports/index.js';
+
+type RebootRestoreCtx = SessionPort & EventPort & ConfigPort & InfraPort;
+
+/** The banner's view of one restorable session. The record itself never leaves the server. */
+function toBannerItem(entry: RebootRestoreEntry) {
+ return {
+ id: entry.sessionId,
+ name: entry.name,
+ workingDir: entry.workingDir,
+ mode: entry.mode,
+ owner: entry.owner,
+ };
+}
+
+export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRestoreCtx): void {
+ const accessorFor = (req: Parameters[0]) => {
+ const user = getAuthUser(req);
+ return (owner: string | undefined) => canAccessOwned(user, owner);
+ };
+
+ // ========== What is on offer ==========
+
+ app.get('/api/reboot-restore', async (req) => {
+ const entries = rebootRestoreRegistry.list(accessorFor(req));
+ return {
+ sessions: entries.map(toBannerItem),
+ // Said plainly here so the banner never implies a full restore: the pane is
+ // new, so the conversation continues and the terminal history does not.
+ scrollbackRestored: false,
+ };
+ });
+
+ // ========== Spend it ==========
+
+ app.post('/api/reboot-restore/restore', async (req, reply) => {
+ const body = parseBody(RebootRestoreRequestSchema, req.body, 'Invalid reboot restore request');
+ const canAccess = accessorFor(req);
+
+ // Take BEFORE the first await: a second click must find nothing to spend.
+ if (!rebootRestoreRegistry.beginSpending()) {
+ return reply.code(409).send(createErrorResponse(ApiErrorCode.CONFLICT, 'A reboot restore is already running'));
+ }
+ const taken = rebootRestoreRegistry.take(canAccess, body.sessionIds);
+
+ try {
+ if (taken.length === 0) return { restored: [], skipped: [] };
+
+ // The plan was built at boot and the board has moved on since. A conversation
+ // the user resumed by hand from the Resume list is already on screen, and a
+ // second pane on it would fight the first for the same transcript.
+ const liveSessionIds = new Set(ctx.sessions.keys());
+ const liveConversationIds = new Set(
+ [...ctx.sessions.values()].map((session) => session.claudeSessionId).filter((id): id is string => !!id)
+ );
+ const { restore, skipped } = rejectAlreadyLive(taken, liveSessionIds, liveConversationIds);
+ // An entry nothing rebuilt stays on offer rather than disappearing silently.
+ rebootRestoreRegistry.restore(skipped.map((s) => taken.find((e) => e.sessionId === s.sessionId)!));
+
+ const restored: ReturnType[] = [];
+ const failures: RebootRestoreRejection[] = [...skipped];
+ const workspaceHooksEnabled = await ctx.getWorkspaceHooksEnabled();
+
+ for (const entry of restore) {
+ // A repo can be deleted between the boot that planned this and the click.
+ if (!existsSync(entry.workingDir)) {
+ failures.push({ sessionId: entry.sessionId, reason: 'workspace-missing' });
+ continue;
+ }
+ try {
+ const saved = entry.state;
+ const claudeModeConfig = await ctx.getClaudeModeConfig();
+ const session = new Session({
+ // The old id is reused on purpose: a pinned record, subagent parents,
+ // window states and the lifecycle log all key off it, and the unpinned
+ // record is gone, so there is nothing to collide with.
+ id: saved.id,
+ workingDir: saved.workingDir,
+ mode: saved.mode,
+ name: saved.name,
+ createdAt: saved.createdAt,
+ mux: ctx.mux,
+ useMux: true,
+ // No `muxSession`: the reboot took the pane with it, so `startInteractive()`
+ // takes its create branch and makes a fresh one.
+ claudeMode: await resolveClaudeModeForUsername(claudeModeConfig.claudeMode, saved.owner),
+ allowedTools: claudeModeConfig.allowedTools,
+ resumeSessionId: entry.resumeConversationId,
+ // Re-resolved against the owner's CURRENT grant, never replayed from the
+ // record: a grant held when the record was written may be gone now.
+ envOverrides: await clampEnvOverridesForOwner(
+ saved.owner,
+ (saved as { __envOverrides?: Record }).__envOverrides
+ ),
+ effort: saved.effort,
+ attachmentHistory:
+ (saved as { __attachmentHistory?: SessionAttachmentHistoryItem[] }).__attachmentHistory ??
+ saved.attachmentHistory,
+ lastSubmitAt: saved.lastSubmitAt,
+ claudeSessionChain: saved.claudeSessionChain,
+ lastActivityAt: saved.lastActivityAt,
+ owner: saved.owner,
+ parentSessionId: saved.parentSessionId,
+ });
+
+ await ctx.addSession(session);
+ ctx.persistSessionState(session);
+ await ctx.setupSessionListeners(session);
+ await session.startInteractive();
+
+ // A session without its workspace hooks goes silently blind: no stop or
+ // idle events for respawn, no Approvals Inbox item, no red tab on a
+ // blocking dialog. The boot-time sweep finished hours ago, so the click
+ // path installs them itself. `hooks: 'always'` is the capability that says
+ // this CLI installs Codeman's hooks into the workspace.
+ if (workspaceHooksEnabled && getCli(session.mode)?.capabilities.hooks === 'always') {
+ await applyWorkspaceHooks(session.workingDir, true).catch((err: unknown) =>
+ console.warn(`[reboot-restore] hook install failed for ${session.workingDir}: ${getErrorMessage(err)}`)
+ );
+ }
+
+ getLifecycleLog().log({ event: 'recovered', sessionId: session.id, name: session.name });
+ // Every other open tab and phone needs this; the clicking tab already has
+ // the response, and the client's handler is an idempotent upsert.
+ ctx.broadcast(SseEvent.SessionCreated, ctx.getSessionStateWithRespawn(session));
+ restored.push(toBannerItem(entry));
+ } catch (err) {
+ // One workspace that has gone missing must not stop the rest of the pass.
+ console.error(`[reboot-restore] failed to rebuild ${entry.sessionId}:`, err);
+ failures.push({ sessionId: entry.sessionId, reason: 'workspace-missing' });
+ }
+ }
+
+ if (restored.length > 0) {
+ // A reboot leaves recovery with nothing alive to find, so its own block never
+ // started the stats collector. This clears and re-arms its interval, so it is
+ // safe to call whether or not the collector is already running.
+ ctx.mux.startStatsCollection(STATS_COLLECTION_INTERVAL_MS);
+ }
+
+ return { restored, skipped: failures };
+ } finally {
+ rebootRestoreRegistry.endSpending();
+ }
+ });
+
+ // ========== Drop it ==========
+
+ app.post('/api/reboot-restore/dismiss', async (req) => {
+ const dismissed = rebootRestoreRegistry.clear(accessorFor(req));
+ return { dismissed };
+ });
+}
diff --git a/src/web/routes/session-routes.ts b/src/web/routes/session-routes.ts
index 03f92e5d..607ec540 100644
--- a/src/web/routes/session-routes.ts
+++ b/src/web/routes/session-routes.ts
@@ -89,6 +89,7 @@ import {
} from '../route-helpers.js';
import { buildAgentCaseMarker, writeAgentCaseMarker } from '../../agent-case-marker.js';
import { canUsernameRunPrivilegedCommands, resolveClaudeModeForUsername } from '../../user-store.js';
+import { clampEnvOverridesForOwner } from '../../session-env-clamp.js';
import { enabledClis, getCli } from '../../config/cli-registry/registry.js';
import { resolveCliLaunchError } from '../../utils/cli-launcher.js';
import { legacyConfigForMode } from '../../session-cli-registry-bridge.js';
@@ -442,72 +443,6 @@ export async function _clampExternalCliBypassForOwner(
};
}
-/**
- * Env-var keys a non-granted owner must not be able to set, because each one
- * hands back privilege the config clamp above just removed, or redirects a
- * credential-resolution endpoint.
- *
- * The DeepSeek three are reachable because `DSH_*` and `DEEPSEEK_*` are
- * allowlisted `envOverrides` prefixes (schemas.ts) — which they have to be, since
- * that is also how a user configures the harness's non-privileged knobs.
- *
- * - `DSH_PERMISSION_MODE` IS the harness's permission switch. Every other CLI's
- * bypass is a command-line FLAG, reachable only through the per-CLI config the
- * clamp already owns; this one is an env var, so the config clamp alone is
- * half a gate.
- * - `DSH_HOME` points the launcher at a profile tree, and a profile's plugin code
- * executes at BOOT, before any approval row can apply. A user who can write a
- * workspace can put a profile in it, so this is the wider of the two.
- * - `DEEPSEEK_BASE_URL` aims the provider endpoint, and `_configureCliEnv()`
- * forwards the SERVER's own `DEEPSEEK_API_KEY` into every dsh pane before
- * `applyEnvOverrides()` runs — so a non-granted owner who could set the base
- * URL would have the operator's API key sent as a bearer credential to a host
- * of their choosing. (`DEEPSEEK_API_KEY` itself stays overridable: supplying
- * your OWN key removes privilege rather than granting it.)
- * - `OMP_AUTH_BROKER_URL`/`OMP_AUTH_BROKER_TOKEN` are where omp resolves
- * credentials from — the same shape as `DEEPSEEK_BASE_URL` above, reachable
- * because `OMP_*` is an allowlisted prefix. Unlike DeepSeek, Codeman does not
- * forward any operator-held key into an omp pane today (omp's provider
- * credentials live in `~/.omp` config files, not env vars), so there is no
- * known concrete exfiltration path yet — clamped defensively anyway, since a
- * non-granted owner redirecting where a shared multi-tenant deployment
- * resolves auth from is not something to allow silently (found in
- * Ark0N/Codeman#353 review; omp's own knobs are otherwise mostly `PI_*`,
- * already allowlisted for pi and not addressed here — see resolveOmpHome()).
- */
-function ownerClampedEnvKeys(): string[] {
- return enabledClis().flatMap((entry) => entry.capabilities.privilegedEnvKeys);
-}
-
-/**
- * Env-var half of the multi-user bypass clamp.
- *
- * `clampExternalCliBypassForOwner()` clamps the per-CLI CONFIG, and for every CLI
- * but DeepSeek that is the whole story. Here it is not: `applyEnvOverrides()` runs
- * AFTER `_configureCliEnv()` in tmux-manager, so an override sent on the SAME
- * request lands last and wins, and a non-granted owner could restore
- * `danger-full-access` on the very request the config clamp downgraded.
- *
- * Keys are DROPPED rather than rewritten: dropping falls through to what
- * `_configureCliEnv()` exports, which is the clamped config and the server's own
- * `DSH_HOME`, i.e. exactly the intended state. No-op in single-user mode and for a
- * granted owner, like every other clamp here
- * (`canUsernameRunPrivilegedCommands()` returns true when `!isMultiUserMode()`),
- * and it returns the caller's own object untouched when there is nothing to strip.
- */
-async function clampEnvOverridesForOwner(
- owner: string | undefined,
- envOverrides: Record | undefined
-): Promise | undefined> {
- if (!envOverrides) return envOverrides;
- const keys = ownerClampedEnvKeys();
- if (!keys.some((key) => key in envOverrides)) return envOverrides;
- if (await canUsernameRunPrivilegedCommands(owner)) return envOverrides;
- const clamped = { ...envOverrides };
- for (const key of keys) delete clamped[key];
- return clamped;
-}
-
/** Test hook: the env-var half of the same multi-user safety gate. */
export const _clampEnvOverridesForOwner = clampEnvOverridesForOwner;
diff --git a/src/web/schemas.ts b/src/web/schemas.ts
index b5dabdd0..f0509f86 100644
--- a/src/web/schemas.ts
+++ b/src/web/schemas.ts
@@ -1161,6 +1161,20 @@ const NotificationEventSchema = z
})
.optional();
+/**
+ * Body of `POST /api/reboot-restore/restore`.
+ *
+ * `sessionIds` restores a subset, and omitting it restores everything the caller
+ * can see. The ids are session ids from `GET /api/reboot-restore`, and an id the
+ * caller does not own is ignored rather than refused, matching how the session
+ * list scopes rather than 403s.
+ */
+export const RebootRestoreRequestSchema = z
+ .object({
+ sessionIds: z.array(z.string().max(128)).max(200).optional(),
+ })
+ .strict();
+
export const SettingsUpdateSchema = z
.object({
// User-facing product branding. This changes browser/UI copy only; package,
diff --git a/src/web/server.ts b/src/web/server.ts
index cf5927c0..065172d5 100644
--- a/src/web/server.ts
+++ b/src/web/server.ts
@@ -39,7 +39,9 @@ import { fileURLToPath } from 'node:url';
import { existsSync, mkdirSync, readFileSync, chmodSync, rmSync, statSync } from 'node:fs';
import fs from 'node:fs/promises';
import { execSync } from 'node:child_process';
-import { hostname as getHostname } from 'node:os';
+import { hostname as getHostname, uptime as osUptime } from 'node:os';
+import { looksLikeHostReboot, newestPersistedActivity, planRebootRestore } from '../reboot-restore.js';
+import { rebootRestoreRegistry } from './reboot-restore-registry.js';
import { dataPath, getDataDir, CODEMAN_INSTANCE } from '../config/instance.js';
import { normalizeBasePath, stripBasePath, joinBasePath } from '../config/base-path.js';
import { GLYPH, palette } from '../cli-style.js';
@@ -171,6 +173,7 @@ import {
registerScheduledRoutes,
registerHookEventRoutes,
registerApprovalRoutes,
+ registerRebootRestoreRoutes,
registerReadMyMindRoutes,
registerStatusTelemetryRoutes,
registerSystemRoutes,
@@ -1060,6 +1063,7 @@ export class WebServer extends EventEmitter {
registerScheduledRoutes(this.app, ctx);
registerHookEventRoutes(this.app, ctx);
registerApprovalRoutes(this.app, ctx);
+ registerRebootRestoreRoutes(this.app, ctx);
registerReadMyMindRoutes(this.app, ctx);
registerStatusTelemetryRoutes(this.app, ctx);
registerSystemRoutes(this.app, ctx);
@@ -2857,6 +2861,52 @@ export class WebServer extends EventEmitter {
return false;
}
+ /**
+ * Work out what a host reboot destroyed, and leave it on offer for the board.
+ *
+ * Runs inside `restoreMuxSessions()`, in the window after `reconcileSessions()`
+ * has reported the dead sessions and before `finalizeRestoredState()` prunes
+ * their records, so `state.json` is still the full picture here. That window is
+ * the only place the plan can be built, which is why the boot pass builds it
+ * even though nothing is rebuilt until a user clicks.
+ *
+ * Nothing is created here. The plan goes to `rebootRestoreRegistry`, the board
+ * offers it as a banner, and `web/routes/reboot-restore-routes` rebuilds what
+ * the user asks for. A wrong reboot guess therefore costs a line of text the
+ * user dismisses, not N CLI processes nobody asked for.
+ *
+ * @returns how many sessions are on offer.
+ */
+ private planRebootRestoreOffer(dead: string[], livePaneCount: number): number {
+ if (dead.length === 0) return 0;
+
+ const persisted = this.store.getSessions();
+ if (
+ !looksLikeHostReboot({
+ livePaneCount,
+ deadSessionCount: dead.length,
+ uptimeSeconds: osUptime(),
+ newestPersistedActivityAt: newestPersistedActivity(persisted),
+ now: Date.now(),
+ })
+ ) {
+ return 0;
+ }
+
+ const { restore, skipped } = planRebootRestore(dead, persisted, (workingDir) => existsSync(workingDir));
+ if (skipped.length > 0) {
+ console.log(`[Server] Reboot restore is passing over ${skipped.length} dead session(s):`);
+ for (const rejection of skipped) {
+ console.log(`[Server] ${rejection.sessionId}: ${rejection.reason}`);
+ }
+ }
+ rebootRestoreRegistry.set(restore);
+ if (restore.length > 0) {
+ console.log(`[Server] Host reboot detected; offering ${restore.length} session(s) for restore`);
+ }
+ return restore.length;
+ }
+
private async restoreMuxSessions(): Promise {
try {
// Reconcile mux sessions to find which ones are still alive (also discovers unknown ones)
@@ -2866,6 +2916,11 @@ export class WebServer extends EventEmitter {
console.log(`[Server] Discovered ${discovered.length} unknown mux session(s)`);
}
+ // Build the reboot-restore offer HERE: `dead` is only known after
+ // reconciliation, and the records it reads are pruned by
+ // `cleanupStaleSessions()` as soon as `finalizeRestoredState()` runs.
+ this.planRebootRestoreOffer(dead, alive.length);
+
if (alive.length > 0 || discovered.length > 0) {
console.log(`[Server] Found ${alive.length + discovered.length} alive mux session(s) from previous run`);
diff --git a/test/reboot-restore.test.ts b/test/reboot-restore.test.ts
new file mode 100644
index 00000000..76506358
--- /dev/null
+++ b/test/reboot-restore.test.ts
@@ -0,0 +1,367 @@
+/**
+ * @fileoverview The decision half of reboot restore, and proof that the existing
+ * recovery construction path can CREATE a resumed pane.
+ *
+ * Three things are under test. `src/reboot-restore.ts` decides whether the
+ * machine rebooted and which dead sessions may be offered back. The plan
+ * registry in `src/web/reboot-restore-registry.ts` holds that offer between the
+ * boot that builds it and the click that spends it. The third is the claim the
+ * whole feature rests on: a `Session` built the way `restoreMuxSessions()`
+ * already builds one, but given no `muxSession` and a `resumeSessionId`, creates
+ * a fresh pane that resumes the old conversation. If that holds, the restore
+ * needs no new session-creation service.
+ *
+ * `reconcileSessions()` reports every session ALIVE under vitest, so the
+ * server's own boot pass cannot be reached from here. The decision logic is
+ * therefore driven directly, and the construction claim is driven through a real
+ * `Session` against the in-memory tmux layer vitest substitutes.
+ */
+import { mkdirSync, rmSync } from 'node:fs';
+import { homedir } from 'node:os';
+import { join } from 'node:path';
+import { afterEach, describe, expect, it } from 'vitest';
+
+import { Session } from '../src/session.js';
+import { TmuxManager } from '../src/tmux-manager.js';
+import type { SessionState } from '../src/types.js';
+import {
+ looksLikeHostReboot,
+ newestPersistedActivity,
+ planRebootRestore,
+ rejectAlreadyLive,
+ resolveResumeConversationId,
+ type RebootRestoreEntry,
+} from '../src/reboot-restore.js';
+import { RebootRestoreRegistry } from '../src/web/reboot-restore-registry.js';
+
+const HOUR = 60 * 60 * 1000;
+const NOW = 1_760_000_000_000;
+
+function persistedSession(overrides: Partial & { id: string }): SessionState {
+ return {
+ pid: 99999,
+ status: 'idle',
+ workingDir: '/tmp/spike',
+ currentTaskId: null,
+ createdAt: NOW - 4 * HOUR,
+ lastActivityAt: NOW - 2 * HOUR,
+ mode: 'claude',
+ ...overrides,
+ } as SessionState;
+}
+
+describe('reboot detection', () => {
+ const base = {
+ livePaneCount: 0,
+ deadSessionCount: 2,
+ // The host came up 10 minutes ago, well after the sessions were last active.
+ uptimeSeconds: 600,
+ newestPersistedActivityAt: NOW - 2 * HOUR,
+ now: NOW,
+ };
+
+ it('calls it a reboot when the socket is empty and the host booted after the last activity', () => {
+ expect(looksLikeHostReboot(base)).toBe(true);
+ });
+
+ it('refuses when some panes survived, which is an ordinary server restart', () => {
+ expect(looksLikeHostReboot({ ...base, livePaneCount: 3 })).toBe(false);
+ });
+
+ it('refuses on a long-uptime host, where someone wiped the tmux socket by hand', () => {
+ // Up for 30 days: the sessions were active long AFTER this boot, so the panes
+ // went away for some reason other than the machine restarting.
+ expect(looksLikeHostReboot({ ...base, uptimeSeconds: 30 * 24 * 60 * 60 })).toBe(false);
+ });
+
+ it('refuses when nothing died', () => {
+ expect(looksLikeHostReboot({ ...base, deadSessionCount: 0 })).toBe(false);
+ });
+
+ it('reads the newest activity stamp across the persisted records', () => {
+ const persisted = {
+ a: persistedSession({ id: 'a', lastActivityAt: NOW - 5 * HOUR }),
+ b: persistedSession({ id: 'b', lastActivityAt: NOW - 1 * HOUR }),
+ };
+ expect(newestPersistedActivity(persisted)).toBe(NOW - 1 * HOUR);
+ });
+});
+
+describe('which dead sessions may be rebuilt', () => {
+ it('rebuilds a session that was simply running when the power went out', () => {
+ const persisted = { live: persistedSession({ id: 'live', status: 'busy' }) };
+ const plan = planRebootRestore(['live'], persisted, () => true);
+ expect(plan.restore.map((s) => s.sessionId)).toEqual(['live']);
+ });
+
+ it('never revives a session the user killed while pinned (COD-142 demotes it to stopped)', () => {
+ const persisted = { killed: persistedSession({ id: 'killed', status: 'stopped', pinned: true }) };
+ const plan = planRebootRestore(['killed'], persisted, () => true);
+ expect(plan.restore).toEqual([]);
+ expect(plan.skipped).toEqual([{ sessionId: 'killed', reason: 'intentionally-ended' }]);
+ });
+
+ it('never revives a session whose record an unpinned kill already deleted', () => {
+ const plan = planRebootRestore(['gone'], {}, () => true);
+ expect(plan.restore).toEqual([]);
+ expect(plan.skipped).toEqual([{ sessionId: 'gone', reason: 'no-persisted-record' }]);
+ });
+
+ it('never revives a pane whose PTY-exit breaker had tripped', () => {
+ const persisted = { crashy: persistedSession({ id: 'crashy', respawnBlocked: true }) };
+ expect(planRebootRestore(['crashy'], persisted, () => true).skipped[0].reason).toBe('respawn-blocked');
+ });
+
+ it('leaves remote sessions to the COD-108 reconnect watcher', () => {
+ const persisted = {
+ r: persistedSession({
+ id: 'r',
+ remote: { hostId: 'h', host: 'example.test', username: 'u', sessionName: 'n', owned: true },
+ } as Partial & { id: string }),
+ };
+ expect(planRebootRestore(['r'], persisted, () => true).skipped[0].reason).toBe('remote-or-docker');
+ });
+
+ it('leaves docker sessions alone, since the container may not be up', () => {
+ const persisted = {
+ d: persistedSession({ id: 'd', docker: { containerId: 'abc', caseId: 'c' } } as Partial & {
+ id: string;
+ }),
+ };
+ expect(planRebootRestore(['d'], persisted, () => true).skipped[0].reason).toBe('remote-or-docker');
+ });
+
+ it('skips a CLI whose history the claude transcript reader does not understand', () => {
+ const persisted = { c: persistedSession({ id: 'c', mode: 'codex' }) };
+ expect(planRebootRestore(['c'], persisted, () => true).skipped[0].reason).toBe('unsupported-mode');
+ });
+});
+
+describe('a workspace that is no longer on disk', () => {
+ it('is kept out of the offer, so a click cannot scaffold a deleted repo', () => {
+ const persisted = { gone: persistedSession({ id: 'gone', workingDir: '/tmp/deleted-repo' }) };
+ const plan = planRebootRestore(['gone'], persisted, () => false);
+ expect(plan.restore).toEqual([]);
+ expect(plan.skipped).toEqual([{ sessionId: 'gone', reason: 'workspace-missing' }]);
+ });
+
+ it('is judged per session, not for the batch', () => {
+ const persisted = {
+ kept: persistedSession({ id: 'kept', workingDir: '/tmp/still-here' }),
+ gone: persistedSession({ id: 'gone', workingDir: '/tmp/deleted-repo' }),
+ };
+ const plan = planRebootRestore(['kept', 'gone'], persisted, (dir) => dir === '/tmp/still-here');
+ expect(plan.restore.map((entry) => entry.sessionId)).toEqual(['kept']);
+ expect(plan.skipped.map((s) => s.reason)).toEqual(['workspace-missing']);
+ });
+});
+
+describe('a conversation that came back on its own before the click', () => {
+ const entry: RebootRestoreEntry = {
+ sessionId: 'abc',
+ workingDir: '/tmp/spike',
+ mode: 'claude',
+ resumeConversationId: 'conv-1',
+ state: persistedSession({ id: 'abc' }),
+ };
+
+ it('is skipped when the user resumed it by hand from the Resume list', () => {
+ // Same conversation, different session id: the Resume list creates a NEW id.
+ const result = rejectAlreadyLive([entry], new Set(['other']), new Set(['conv-1']));
+ expect(result.restore).toEqual([]);
+ expect(result.skipped).toEqual([{ sessionId: 'abc', reason: 'already-live' }]);
+ });
+
+ it('is skipped when a session with that id is already on the board', () => {
+ const result = rejectAlreadyLive([entry], new Set(['abc']), new Set());
+ expect(result.skipped).toEqual([{ sessionId: 'abc', reason: 'already-live' }]);
+ });
+
+ it('is rebuilt when neither its id nor its conversation is live', () => {
+ const result = rejectAlreadyLive([entry], new Set(['other']), new Set(['conv-other']));
+ expect(result.restore.map((e) => e.sessionId)).toEqual(['abc']);
+ expect(result.skipped).toEqual([]);
+ });
+});
+
+describe('the plan the banner spends', () => {
+ const all = () => true;
+ const entryFor = (sessionId: string, owner?: string): RebootRestoreEntry => ({
+ sessionId,
+ owner,
+ workingDir: '/tmp/spike',
+ mode: 'claude',
+ resumeConversationId: `conv-${sessionId}`,
+ state: persistedSession({ id: sessionId, owner }),
+ });
+
+ it('hands an entry to the first caller and nothing to the second', () => {
+ const registry = new RebootRestoreRegistry();
+ registry.set([entryFor('a'), entryFor('b')]);
+ expect(registry.take(all).map((e) => e.sessionId)).toEqual(['a', 'b']);
+ // The double-click: two panes on one conversation is what this prevents.
+ expect(registry.take(all)).toEqual([]);
+ });
+
+ it('spends only the ids a caller asked for', () => {
+ const registry = new RebootRestoreRegistry();
+ registry.set([entryFor('a'), entryFor('b')]);
+ expect(registry.take(all, ['b']).map((e) => e.sessionId)).toEqual(['b']);
+ expect(registry.list(all).map((e) => e.sessionId)).toEqual(['a']);
+ });
+
+ it("shows a user their own sessions and leaves another owner's alone", () => {
+ const registry = new RebootRestoreRegistry();
+ registry.set([entryFor('mine', 'alice'), entryFor('theirs', 'bob')]);
+ const asAlice = (owner: string | undefined) => owner === 'alice';
+ expect(registry.list(asAlice).map((e) => e.sessionId)).toEqual(['mine']);
+ expect(registry.take(asAlice).map((e) => e.sessionId)).toEqual(['mine']);
+ // Bob's entry is still on offer for Bob.
+ expect(registry.list(() => true).map((e) => e.sessionId)).toEqual(['theirs']);
+ });
+
+ it('puts back an entry that no pane was created for', () => {
+ const registry = new RebootRestoreRegistry();
+ registry.set([entryFor('a')]);
+ const taken = registry.take(all);
+ registry.restore(taken);
+ expect(registry.list(all).map((e) => e.sessionId)).toEqual(['a']);
+ });
+
+ it('runs one restore at a time', () => {
+ const registry = new RebootRestoreRegistry();
+ expect(registry.beginSpending()).toBe(true);
+ expect(registry.beginSpending()).toBe(false);
+ registry.endSpending();
+ expect(registry.beginSpending()).toBe(true);
+ });
+
+ it('drops what a dismiss cleared', () => {
+ const registry = new RebootRestoreRegistry();
+ registry.set([entryFor('a'), entryFor('b')]);
+ expect(registry.clear(all)).toBe(2);
+ expect(registry.list(all)).toEqual([]);
+ });
+
+ it('forgets a plan nobody took for a day', () => {
+ const registry = new RebootRestoreRegistry();
+ registry.set([entryFor('a')]);
+ const dayLater = Date.now() + 25 * HOUR;
+ const realNow = Date.now;
+ Date.now = () => dayLater;
+ try {
+ expect(registry.list(all)).toEqual([]);
+ } finally {
+ Date.now = realNow;
+ }
+ });
+});
+
+describe('which conversation a rebuilt pane resumes', () => {
+ it('prefers the chain tail, the conversation the CLI reported last', () => {
+ const state = persistedSession({
+ id: 'sess-1',
+ resumeSessionId: 'launch-id',
+ claudeSessionChain: ['launch-id', 'after-clear'],
+ });
+ expect(resolveResumeConversationId(state)).toBe('after-clear');
+ });
+
+ it('falls back to the id the session originally resumed', () => {
+ const state = persistedSession({ id: 'sess-1', resumeSessionId: 'resumed-id' });
+ expect(resolveResumeConversationId(state)).toBe('resumed-id');
+ });
+
+ it('falls back to the session id, which is what Claude was launched with', () => {
+ expect(resolveResumeConversationId(persistedSession({ id: 'sess-1' }))).toBe('sess-1');
+ });
+});
+
+describe('the recovery construction path can create a resumed pane', () => {
+ const workingDir = join(homedir(), 'codeman-cases', 'reboot-restore-spike');
+ const sessions: Session[] = [];
+
+ afterEach(() => {
+ for (const s of sessions.splice(0)) s.stop();
+ rmSync(workingDir, { recursive: true, force: true });
+ });
+
+ /** Built exactly as the reboot pass builds one: no `muxSession`, plus a resume id. */
+ function rebuildFromPersistedState(state: SessionState, mux: TmuxManager): Session {
+ mkdirSync(workingDir, { recursive: true });
+ const session = new Session({
+ id: state.id,
+ workingDir,
+ mode: state.mode,
+ name: state.name,
+ createdAt: state.createdAt,
+ mux,
+ useMux: true,
+ resumeSessionId: resolveResumeConversationId(state),
+ owner: state.owner,
+ lastActivityAt: state.lastActivityAt,
+ claudeSessionChain: state.claudeSessionChain,
+ });
+ sessions.push(session);
+ return session;
+ }
+
+ it('creates a NEW mux session rather than needing one to attach to', async () => {
+ const mux = new TmuxManager();
+ const state = persistedSession({ id: 'aaaaaaa1-1111-4111-8111-111111111111', name: 'w1-spike' });
+ const session = rebuildFromPersistedState(state, mux);
+
+ expect(mux.getSessions()).toHaveLength(0);
+ await session.startInteractive();
+
+ const created = mux.getSessions();
+ expect(created).toHaveLength(1);
+ expect(created[0].sessionId).toBe('aaaaaaa1-1111-4111-8111-111111111111');
+ expect(created[0].workingDir).toBe(workingDir);
+ });
+
+ it('comes back pointed at the conversation the pane was holding', async () => {
+ const mux = new TmuxManager();
+ const state = persistedSession({
+ id: 'aaaaaaa2-2222-4222-8222-222222222222',
+ resumeSessionId: 'launch-id',
+ claudeSessionChain: ['launch-id', 'after-clear'],
+ });
+ const session = rebuildFromPersistedState(state, mux);
+
+ await session.startInteractive();
+
+ // The chain tail wins: a `/clear` before the reboot moved the CLI off the launch id.
+ expect(session.claudeSessionId).toBe('after-clear');
+ });
+
+ it('comes back idle, with no prompt sent and no autonomous loop armed', async () => {
+ const mux = new TmuxManager();
+ const state = persistedSession({
+ id: 'aaaaaaa3-3333-4333-8333-333333333333',
+ ralphEnabled: true,
+ respawnEnabled: true,
+ });
+ const session = rebuildFromPersistedState(state, mux);
+
+ await session.startInteractive();
+
+ // No prompt was queued: nothing is waiting on a task. The status itself is not
+ // assertable here, because the test PTY echoes and the activity detector reads
+ // that echo as work; in production the pane settles once the CLI finishes booting.
+ expect(session.currentTaskId).toBeNull();
+ // The pass never touches the tracker, so a persisted Ralph loop stays cold.
+ expect(session.ralphTracker.enabled).toBe(false);
+ });
+
+ it('keeps the owner it was persisted with, there being no request to read one from', async () => {
+ const mux = new TmuxManager();
+ const state = persistedSession({ id: 'aaaaaaa4-4444-4444-8444-444444444444', owner: 'alice' });
+ const session = rebuildFromPersistedState(state, mux);
+
+ await session.startInteractive();
+
+ expect(session.owner).toBe('alice');
+ expect(mux.getSessions()[0].owner).toBe('alice');
+ });
+});
diff --git a/test/routes/reboot-restore-routes.test.ts b/test/routes/reboot-restore-routes.test.ts
new file mode 100644
index 00000000..c16cddac
--- /dev/null
+++ b/test/routes/reboot-restore-routes.test.ts
@@ -0,0 +1,175 @@
+/**
+ * Reboot-restore route tests (src/web/routes/reboot-restore-routes.ts) via
+ * app.inject(), no live port.
+ *
+ * Every entry these tests put on offer names a workspace that does not exist, so
+ * the route's click-time workspace check rejects it before any `Session` is
+ * constructed. That keeps the tests on the route's own guards — taking, scoping,
+ * single-flighting and re-checking — and leaves pane creation to
+ * test/reboot-restore.test.ts, which drives a real `Session` for it.
+ *
+ * The routes read the process-wide `rebootRestoreRegistry` singleton, so every
+ * test resets it; a leaked entry would bleed into the next one.
+ */
+import { describe, it, expect, afterEach } from 'vitest';
+import Fastify, { type FastifyInstance } from 'fastify';
+import fastifyCookie from '@fastify/cookie';
+import { registerRebootRestoreRoutes } from '../../src/web/routes/reboot-restore-routes.js';
+import { rebootRestoreRegistry } from '../../src/web/reboot-restore-registry.js';
+import { installRouteErrorHandler } from '../../src/web/route-error-handler.js';
+import { httpStatusForErrorCode, type ApiErrorCode } from '../../src/types.js';
+import { createMockRouteContext } from '../mocks/index.js';
+import type { RebootRestoreEntry } from '../../src/reboot-restore.js';
+import type { SessionState } from '../../src/types.js';
+
+async function createHarness(authUser?: { username: string; role: 'admin' | 'user' }): Promise {
+ const app = Fastify({ logger: false });
+ await app.register(fastifyCookie);
+ if (authUser) {
+ app.addHook('onRequest', async (req) => {
+ (req as unknown as { authUser: typeof authUser }).authUser = authUser;
+ });
+ }
+ registerRebootRestoreRoutes(app, createMockRouteContext() as never);
+
+ app.addHook('preSerialization', (req, reply, payload: unknown, done) => {
+ if (!req.url.startsWith('/api')) return done(null, payload);
+ if (payload === null || typeof payload !== 'object') return done(null, payload);
+ const p = payload as { success?: unknown; errorCode?: unknown };
+ if (p.success === false) {
+ if (reply.statusCode === 200 && typeof p.errorCode === 'string') {
+ reply.code(httpStatusForErrorCode(p.errorCode as ApiErrorCode));
+ }
+ return done(null, payload);
+ }
+ if (p.success === true) return done(null, payload);
+ return done(null, { success: true, data: payload });
+ });
+
+ installRouteErrorHandler(app);
+ await app.ready();
+ return app;
+}
+
+/** An entry whose workspace is deliberately absent, so no pane is ever created. */
+function offerEntry(sessionId: string, owner?: string): RebootRestoreEntry {
+ return {
+ sessionId,
+ name: `session ${sessionId}`,
+ workingDir: `/tmp/codeman-reboot-restore-missing/${sessionId}`,
+ owner,
+ mode: 'claude',
+ resumeConversationId: `conv-${sessionId}`,
+ state: {
+ id: sessionId,
+ pid: null,
+ status: 'idle',
+ workingDir: `/tmp/codeman-reboot-restore-missing/${sessionId}`,
+ currentTaskId: null,
+ createdAt: 1_760_000_000_000,
+ mode: 'claude',
+ owner,
+ } as SessionState,
+ };
+}
+
+afterEach(() => {
+ rebootRestoreRegistry.reset();
+});
+
+describe('GET /api/reboot-restore', () => {
+ it('reports nothing when no reboot left anything behind', async () => {
+ const app = await createHarness();
+ const res = await app.inject({ method: 'GET', url: '/api/reboot-restore' });
+ expect(res.statusCode).toBe(200);
+ expect(res.json().data.sessions).toEqual([]);
+ await app.close();
+ });
+
+ it('names what is on offer, and says the scrollback is not coming back', async () => {
+ rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]);
+ const app = await createHarness();
+ const body = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data;
+ expect(body.sessions.map((s: { id: string }) => s.id)).toEqual(['a', 'b']);
+ expect(body.scrollbackRestored).toBe(false);
+ await app.close();
+ });
+
+ it('never carries the persisted record itself to the browser', async () => {
+ rebootRestoreRegistry.set([offerEntry('a', 'alice')]);
+ const app = await createHarness({ username: 'alice', role: 'admin' });
+ const body = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data;
+ expect(Object.keys(body.sessions[0]).sort()).toEqual(['id', 'mode', 'name', 'owner', 'workingDir']);
+ expect(body.sessions[0].state).toBeUndefined();
+ await app.close();
+ });
+});
+
+describe('POST /api/reboot-restore/restore', () => {
+ it('spends the offer, so a second click finds nothing left to spend', async () => {
+ rebootRestoreRegistry.set([offerEntry('a')]);
+ const app = await createHarness();
+
+ const first = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data;
+ // The workspace is gone, so nothing was rebuilt — but the entry was taken.
+ expect(first.restored).toEqual([]);
+ expect(first.skipped).toEqual([{ sessionId: 'a', reason: 'workspace-missing' }]);
+
+ const second = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data;
+ expect(second.restored).toEqual([]);
+ expect(second.skipped).toEqual([]);
+ await app.close();
+ });
+
+ it('spends only the sessions the click named', async () => {
+ rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]);
+ const app = await createHarness();
+
+ const res = await app.inject({
+ method: 'POST',
+ url: '/api/reboot-restore/restore',
+ payload: { sessionIds: ['b'] },
+ });
+ expect(res.json().data.skipped).toEqual([{ sessionId: 'b', reason: 'workspace-missing' }]);
+
+ const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data;
+ expect(left.sessions.map((s: { id: string }) => s.id)).toEqual(['a']);
+ await app.close();
+ });
+
+ it('refuses a body it does not recognise rather than guessing', async () => {
+ const app = await createHarness();
+ const res = await app.inject({
+ method: 'POST',
+ url: '/api/reboot-restore/restore',
+ payload: { sessionIds: 'not-an-array' },
+ });
+ expect(res.statusCode).toBeGreaterThanOrEqual(400);
+ await app.close();
+ });
+
+ it('turns a second concurrent restore away rather than interleaving it', async () => {
+ rebootRestoreRegistry.set([offerEntry('a')]);
+ // Claimed by a restore already in flight.
+ expect(rebootRestoreRegistry.beginSpending()).toBe(true);
+ const app = await createHarness();
+ const res = await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} });
+ expect(res.statusCode).toBe(409);
+ rebootRestoreRegistry.endSpending();
+ await app.close();
+ });
+});
+
+describe('POST /api/reboot-restore/dismiss', () => {
+ it('drops the offer and leaves the banner with nothing to show', async () => {
+ rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]);
+ const app = await createHarness();
+
+ const res = await app.inject({ method: 'POST', url: '/api/reboot-restore/dismiss', payload: {} });
+ expect(res.json().data.dismissed).toBe(2);
+
+ const after = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data;
+ expect(after.sessions).toEqual([]);
+ await app.close();
+ });
+});
From fbede5cd2a20bfc50074c113da7247adb09adf0c Mon Sep 17 00:00:00 2001
From: Michael Grundberg
Date: Wed, 16 Sep 2026 11:44:37 +0200
Subject: [PATCH 11/28] fix(sessions): act on the dual review of the
reboot-restore route
Fifteen findings from two independent reviews of #442, three of them
blocking. Every one is addressed here.
The three blockers all sat in the restore route. A rebuild that threw after
addSession left a registered session with no pane behind it, visible on the
board, holding a layout slot and written to state.json, with its plan entry
already spent; the catch now cleans the session up and puts the entry back.
The loop checked neither the global nor the per-user session cap, so one
click could take a board past a documented limit; capacity is now re-checked
per iteration, because the loop is itself creating the sessions it counts.
Worst of the three, a rebuilt session carried none of the state its
constructor has no parameter for and then persisted itself over the record
that held it, zeroing token and cost totals and dropping the pin. The pin
matters most: pruning keeps a record only while it is pinned, so discarding
it handed the record to the next stale sweep. A new
reapplyPersistedSessionState() on the session port restores the pin, the
token totals, auto-compact, auto-clear, auto-resume, nice priority, the
flicker filter and the custom-model selection, and it runs before both
startInteractive and the first persist.
The rest, in the order they bite a user. Every rebuild failure was reported
as workspace-missing, so the banner told users their repo was gone when the
agent had simply failed to start; there are now distinct reasons, and the
toast names each one. The client read restored and skipped off the outer
response object rather than through the uniform envelope, so every count
came back zero and neither toast ever fired. A board left open across the
reboot never learned an offer existed, because the banner was seeded only on
the page-load path; it now re-reads on every SSE init. The workspace check
was existence-only, skipping the multi-user confinement that the create
route applies, so a withdrawn grant would not be noticed. The banner had no
phone breakpoint while its text was nowrap and its buttons could not shrink.
Smaller: a missing workspace is now re-offered rather than dropped, while an
already-open conversation is dropped rather than re-offered forever; a throw
anywhere in the route returns the unspent entries instead of discarding the
plan; the single flight is keyed by owner, since take() already stops two
callers receiving one entry; the env clamp's header no longer claims a
protection it cannot provide on this path today, and names the check that
does bite; the three endpoints are documented in docs/api-reference.md; and
the module header now says that os.uptime() reads the host's clock, so the
feature is effectively off inside a container.
The review also explained why the tests missed all of this: they proved the
construction claim through their own copy of the construction rather than
through the route, and the route tests used workspaces that did not exist,
so no Session was ever built. test/routes/reboot-restore-rebuild-failure.ts
mocks the Session module to drive the route's real path, and covers the
cleanup, the reason reported, the re-application ordering, the broadcast and
the caps. The mock route context gains the port method and the mux call the
route needs.
Refs #411
Co-Authored-By: Claude Opus 5 (1M context)
---
docs/api-reference.md | 201 ++++++++++-------
src/reboot-restore.ts | 20 +-
src/session-env-clamp.ts | 14 +-
src/web/ports/session-port.ts | 12 +
src/web/public/app.js | 10 +-
src/web/public/mobile.css | 35 +++
src/web/public/reboot-restore-ui.js | 43 +++-
src/web/reboot-restore-registry.ts | 39 ++--
src/web/routes/reboot-restore-routes.ts | 72 +++++-
src/web/server.ts | 48 ++++
test/mocks/mock-route-context.ts | 2 +
.../reboot-restore-rebuild-failure.test.ts | 210 ++++++++++++++++++
test/routes/reboot-restore-routes.test.ts | 47 +++-
13 files changed, 632 insertions(+), 121 deletions(-)
create mode 100644 test/routes/reboot-restore-rebuild-failure.test.ts
diff --git a/docs/api-reference.md b/docs/api-reference.md
index dcd0d631..85349ee3 100644
--- a/docs/api-reference.md
+++ b/docs/api-reference.md
@@ -66,17 +66,17 @@ The single source of truth is `ErrorStatus` / `httpStatusForErrorCode()` in
`src/types/api.ts`. Clients should branch on `errorCode` (stable) and may rely on
the HTTP status.
-| `errorCode` | HTTP | Meaning |
-|-------------|------|---------|
-| `INVALID_INPUT` | 400 | Malformed request / failed validation |
-| `UNAUTHORIZED` | 401 | Authentication required or failed |
-| `NOT_FOUND` | 404 | Resource does not exist |
-| `SESSION_BUSY` | 409 | Session is busy |
-| `CONFLICT` | 409 | Conflicts with current state (e.g. already running) |
-| `ALREADY_EXISTS` | 409 | Resource already exists |
-| `OPERATION_FAILED` | 422 | Well-formed but could not be completed |
-| `RATE_LIMITED` | 429 | Too many requests |
-| `INTERNAL_ERROR` | 500 | Unexpected server error |
+| `errorCode` | HTTP | Meaning |
+| ------------------ | ---- | --------------------------------------------------- |
+| `INVALID_INPUT` | 400 | Malformed request / failed validation |
+| `UNAUTHORIZED` | 401 | Authentication required or failed |
+| `NOT_FOUND` | 404 | Resource does not exist |
+| `SESSION_BUSY` | 409 | Session is busy |
+| `CONFLICT` | 409 | Conflicts with current state (e.g. already running) |
+| `ALREADY_EXISTS` | 409 | Resource already exists |
+| `OPERATION_FAILED` | 422 | Well-formed but could not be completed |
+| `RATE_LIMITED` | 429 | Too many requests |
+| `INTERNAL_ERROR` | 500 | Unexpected server error |
Adding a new error code is non-breaking; removing or renaming one is a major change.
@@ -87,10 +87,10 @@ exist because SSE is Codeman's only other "tell me when" channel, and an agent
driving the API from a shell tool cannot practically hold a stream and parse
events inline.
-| Call | Blocks until |
-|------|--------------|
-| `GET /api/v1/sessions/:id/wait` | one of a set of lifecycle signals fires |
-| `GET /api/v1/sessions/:id/wait-output` | a literal string appears in the session's output |
+| Call | Blocks until |
+| --------------------------------------------- | -------------------------------------------------- |
+| `GET /api/v1/sessions/:id/wait` | one of a set of lifecycle signals fires |
+| `GET /api/v1/sessions/:id/wait-output` | a literal string appears in the session's output |
| `POST /api/v1/sessions/:id/input` with `wait` | the input is delivered **and then** a signal fires |
`POST .../input` with `wait` is not the same as a `POST` followed by a separate
@@ -140,13 +140,13 @@ contract is a **marker unique to each call** (`MARK="DONE_$RANDOM"`, send
### Signals
-| Signal | Source | Actually fires for |
-|--------|--------|--------------------|
-| `idle` | the session's own `idle` event | `claude`: yes, on ❯-prompt detection after activity. `shell`: **once only**, ~500 ms after start, and never again. External CLIs: not guaranteed (they render their own TUIs and readiness is output stabilization) |
-| `working` | the session's own `working` event | `claude` only in practice (spinner and work-keyword detection are Claude output formats) |
-| `stop` | the Claude Code `stop` hook, the definitive end-of-turn signal | `claude` only |
-| `blocked` | a `permission_prompt` or `elicitation_dialog` hook | `claude` only, and rarer than it looks: see below |
-| `exit` | no process is behind the session | every mode |
+| Signal | Source | Actually fires for |
+| --------- | -------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `idle` | the session's own `idle` event | `claude`: yes, on ❯-prompt detection after activity. `shell`: **once only**, ~500 ms after start, and never again. External CLIs: not guaranteed (they render their own TUIs and readiness is output stabilization) |
+| `working` | the session's own `working` event | `claude` only in practice (spinner and work-keyword detection are Claude output formats) |
+| `stop` | the Claude Code `stop` hook, the definitive end-of-turn signal | `claude` only |
+| `blocked` | a `permission_prompt` or `elicitation_dialog` hook | `claude` only, and rarer than it looks: see below |
+| `exit` | no process is behind the session | every mode |
`stop` is the signal to orchestrate on where it exists; `idle` is a heuristic
fallback that can flap mid-turn when a spinner pauses. The default set when `until`
@@ -156,12 +156,12 @@ can no longer happen). On a `claude` worker, prefer an explicit `until=stop,exit
once the session is up: the default set's `idle` also resolves on a spinner pause,
and on a fresh session the **startup** `idle` (emitted when the CLI first comes up)
can land inside your first wait window and report a turn that never ran. Measured:
-a session parked on the trust dialog emits no *further* `idle`, so it is the
+a session parked on the trust dialog emits no _further_ `idle`, so it is the
startup transition, not the dialog, that produces the false success below.
⚠️ **`exit` means "nothing is running", which includes "not started yet".** The
server answers from `pid === null` plus a mux-layer pane-death probe, and that
-covers a session that exited — including a worker that died *inside* its tmux pane
+covers a session that exited — including a worker that died _inside_ its tmux pane
while the local attach client (and therefore `pid`) lives on — one that was
detached, and one that was **created but never started**. So the first wait
after `POST /api/v1/sessions` returns `{"signal":"exit","immediate":true}` in
@@ -184,7 +184,7 @@ blocked, and polling `blocked` alone will sit at its timeout.
⚠️ **On a `shell` session, only `exit` and marker-matching are dependable.** A shell
session emits its one `idle` at startup and then stays `status: "idle"` forever,
-whatever the pane is doing, so it never emits a *transition*. Since send-and-wait
+whatever the pane is doing, so it never emits a _transition_. Since send-and-wait
requires a transition (and so does `fresh=1`), both can only time out there:
a documented default `wait` on a shell worker running `sleep 4` times out at the
full 25 s. Synchronize hook-less sessions with `wait-output` and a unique marker
@@ -218,11 +218,11 @@ with `from=buffer` keeps matching long after the dialog is gone. A worked versio
### `GET /api/v1/sessions/:id/wait`
-| Param | Type | Default | Notes |
-|-------|------|---------|-------|
-| `until` | comma-separated list of `idle,working,stop,blocked,exit` | `stop,idle,exit` | resolves on the first to fire. An unknown token is a `400` naming it, never a silent fallback |
-| `timeout` | positive integer ms | `60000` | **validated first, clamped second.** `0`, a negative value and a fractional value are all `400`s, not clamps; a valid value outside `[1000, 600000]` is clamped and echoed as `wait.timeoutMs` |
-| `fresh` | `0` \| `1` \| `false` \| `true` | `0` | `1` requires an actual transition, ignoring the state at call time |
+| Param | Type | Default | Notes |
+| --------- | -------------------------------------------------------- | ---------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `until` | comma-separated list of `idle,working,stop,blocked,exit` | `stop,idle,exit` | resolves on the first to fire. An unknown token is a `400` naming it, never a silent fallback |
+| `timeout` | positive integer ms | `60000` | **validated first, clamped second.** `0`, a negative value and a fractional value are all `400`s, not clamps; a valid value outside `[1000, 600000]` is clamped and echoed as `wait.timeoutMs` |
+| `fresh` | `0` \| `1` \| `false` \| `true` | `0` | `1` requires an actual transition, ignoring the state at call time |
```bash
curl -s "$API/api/v1/sessions/$SID/wait?until=stop,exit&timeout=60000"
@@ -239,12 +239,12 @@ a plain signal wait, so check the endpoint path before blaming the parameters.
### `GET /api/v1/sessions/:id/wait-output`
-| Param | Type | Default | Notes |
-|-------|------|---------|-------|
-| `match` | literal string, 1 to 200 chars | required | substring match against the PTY stream with ANSI escapes stripped. A match spanning two PTY chunks is found |
-| `nocase` | `0` \| `1` \| `false` \| `true` | `0` | case-insensitive compare. The returned snippet keeps the terminal's original casing |
-| `from` | `now` \| `buffer` | `now` | `buffer` scans the tail of the existing terminal buffer (bounded, 256 KB by default) before blocking |
-| `timeout` | positive integer ms | `60000` | same validation and clamp as `/wait` |
+| Param | Type | Default | Notes |
+| --------- | ------------------------------- | -------- | ----------------------------------------------------------------------------------------------------------- |
+| `match` | literal string, 1 to 200 chars | required | substring match against the PTY stream with ANSI escapes stripped. A match spanning two PTY chunks is found |
+| `nocase` | `0` \| `1` \| `false` \| `true` | `0` | case-insensitive compare. The returned snippet keeps the terminal's original casing |
+| `from` | `now` \| `buffer` | `now` | `buffer` scans the tail of the existing terminal buffer (bounded, 256 KB by default) before blocking |
+| `timeout` | positive integer ms | `60000` | same validation and clamp as `/wait` |
**Matching is literal, never a pattern.** A `regex` parameter is rejected with a
`400` rather than ignored, so a caller that assumed otherwise finds out immediately
@@ -296,10 +296,10 @@ hand-written query string decodes to a space.
Two optional fields on the existing endpoint:
-| Field | Type | Notes |
-|-------|------|-------|
-| `wait` | `true` or the same comma grammar as `until` | `true` means the default signal set. Omitted keeps the historical fire-and-forget behavior, unchanged. `null`, `false` and an empty string are all read as **absent**, not as an error and not as "wait for the default" |
-| `waitTimeout` | positive integer ms | same validation **and** clamp as `timeout`: `0`, a negative and a fractional value are `400`s, anything valid is clamped into `[1000, 600000]` and echoed as `wait.timeoutMs` |
+| Field | Type | Notes |
+| ------------- | ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
+| `wait` | `true` or the same comma grammar as `until` | `true` means the default signal set. Omitted keeps the historical fire-and-forget behavior, unchanged. `null`, `false` and an empty string are all read as **absent**, not as an error and not as "wait for the default" |
+| `waitTimeout` | positive integer ms | same validation **and** clamp as `timeout`: `0`, a negative and a fractional value are `400`s, anything valid is clamped into `[1000, 600000]` and echoed as `wait.timeoutMs` |
Both are `nullish`, so an explicit `null` from `JSON.stringify` is accepted as
"absent" rather than failing validation. That is deliberate: `.optional()` would
@@ -330,16 +330,24 @@ All three nest the wait result under `data.wait`, so one client helper works aga
any of them:
```json
-{ "success": true, "data": {
- "sessionId": "28325fd3-caa7-4178-82bf-87dfebf0f464",
- "status": "idle",
- "limitPaused": false,
- "wait": {
- "signal": "stop", "until": ["stop", "idle", "exit"],
- "timedOut": false, "immediate": false, "ended": false, "aborted": false,
- "waitedMs": 8421, "timeoutMs": 60000
+{
+ "success": true,
+ "data": {
+ "sessionId": "28325fd3-caa7-4178-82bf-87dfebf0f464",
+ "status": "idle",
+ "limitPaused": false,
+ "wait": {
+ "signal": "stop",
+ "until": ["stop", "idle", "exit"],
+ "timedOut": false,
+ "immediate": false,
+ "ended": false,
+ "aborted": false,
+ "waitedMs": 8421,
+ "timeoutMs": 60000
+ }
}
-}}
+}
```
`POST .../input` returns the same `wait` object alongside `delivered`, `duplicate`,
@@ -353,21 +361,21 @@ redelivery (harmless, the turn it refers to may be long over), while with
client that reads `delivered === false` as "duplicate" silently treats a failed send
as a success.
-| Field | Type | Meaning |
-|-------|------|---------|
-| `wait.signal` | signal \| `null` | the signal that fired (`/wait` and `/input` only) |
-| `wait.until` | array of signals | what the server actually waited on, after narrowing the default set for the session's mode (`/wait` and `/input` only) |
-| `wait.matched` | boolean | the string appeared (`/wait-output` only) |
-| `wait.match` | string | the literal that was searched for (`/wait-output` only) |
-| `wait.snippet` | string \| `null` | bounded window of output around the match, blank runs collapsed for readability (`/wait-output` only) |
-| `wait.timedOut` | boolean | the wait hit its timeout. Still a `200` |
-| `wait.immediate` | boolean | the condition already held at call time, so nothing was waited for (`waitedMs` is 0) |
-| `wait.ended` | boolean | the session went away (deleted or torn down) before the condition was met |
-| `wait.aborted` | boolean | the client hung up, so the waiter was released without resolving — and by that definition a client never reads `true`. When the **server** abandons a wait itself (send-and-wait against a session with no PTY), it answers in about a millisecond with `ended: true`, `delivered: false`, `duplicate: false` and `aborted: false`: `delivered`/`ended` carry that story, and `aborted` stays the transport flag. Present for completeness; treat a `true` as "this wait answered nothing", never as an outcome |
-| `wait.waitedMs` | number | wall-clock ms actually spent waiting |
-| `wait.timeoutMs` | number | the timeout **after clamping**, which is what was applied |
-| `status` | `SessionStatus` | the session's status after the wait, so a caller that timed out still learns where things stand |
-| `limitPaused` | boolean | the session is paused on a usage limit and will emit nothing until its reset, so a timeout here is expected rather than a stall worth retrying hard |
+| Field | Type | Meaning |
+| ---------------- | ---------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `wait.signal` | signal \| `null` | the signal that fired (`/wait` and `/input` only) |
+| `wait.until` | array of signals | what the server actually waited on, after narrowing the default set for the session's mode (`/wait` and `/input` only) |
+| `wait.matched` | boolean | the string appeared (`/wait-output` only) |
+| `wait.match` | string | the literal that was searched for (`/wait-output` only) |
+| `wait.snippet` | string \| `null` | bounded window of output around the match, blank runs collapsed for readability (`/wait-output` only) |
+| `wait.timedOut` | boolean | the wait hit its timeout. Still a `200` |
+| `wait.immediate` | boolean | the condition already held at call time, so nothing was waited for (`waitedMs` is 0) |
+| `wait.ended` | boolean | the session went away (deleted or torn down) before the condition was met |
+| `wait.aborted` | boolean | the client hung up, so the waiter was released without resolving — and by that definition a client never reads `true`. When the **server** abandons a wait itself (send-and-wait against a session with no PTY), it answers in about a millisecond with `ended: true`, `delivered: false`, `duplicate: false` and `aborted: false`: `delivered`/`ended` carry that story, and `aborted` stays the transport flag. Present for completeness; treat a `true` as "this wait answered nothing", never as an outcome |
+| `wait.waitedMs` | number | wall-clock ms actually spent waiting |
+| `wait.timeoutMs` | number | the timeout **after clamping**, which is what was applied |
+| `status` | `SessionStatus` | the session's status after the wait, so a caller that timed out still learns where things stand |
+| `limitPaused` | boolean | the session is paused on a usage limit and will emit nothing until its reset, so a timeout here is expected rather than a stall worth retrying hard |
Read the outcome by discriminator, in this order:
@@ -390,12 +398,12 @@ read the timeout as "the worker is wedged" and kill a session that was working f
### Errors
-| `errorCode` | HTTP | When |
-|-------------|------|------|
-| `INVALID_INPUT` | 400 | unknown `until` / `wait` token; `stop` or `blocked` requested explicitly on a mode that installs no hooks (the message names the mode); `regex=` on `/wait-output`; `match` outside 1 to 200 chars; a non-numeric `timeout` |
-| `NOT_FOUND` | 404 | no such session, or one this caller does not own |
-| `SESSION_BUSY` | 409 | this session's waiter cap is full |
-| `RATE_LIMITED` | 429 | a per-owner or process-wide waiter cap is full. Retry later; the session you named is not the problem |
+| `errorCode` | HTTP | When |
+| --------------- | ---- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `INVALID_INPUT` | 400 | unknown `until` / `wait` token; `stop` or `blocked` requested explicitly on a mode that installs no hooks (the message names the mode); `regex=` on `/wait-output`; `match` outside 1 to 200 chars; a non-numeric `timeout` |
+| `NOT_FOUND` | 404 | no such session, or one this caller does not own |
+| `SESSION_BUSY` | 409 | this session's waiter cap is full |
+| `RATE_LIMITED` | 429 | a per-owner or process-wide waiter cap is full. Retry later; the session you named is not the problem |
The two capacity codes are deliberately different. A process-wide cap reported as
`SESSION_BUSY` would tell the caller to switch sessions, which cannot help. The
@@ -446,9 +454,9 @@ Design: [`approvals-inbox-plan.md`](approvals-inbox-plan.md).
- `GET /api/v1/approvals` → `{ approvals: ApprovalItem[] }`, oldest first,
ownership-scoped in multi-user mode. `ApprovalItem`: `{ id, sessionId,
- sessionName, kind: 'permission'|'question'|'idle', createdAt, toolName?,
- toolSummary?, message?, cwd?, context?, options?: {n, label}[],
- acknowledgedAt? }`. `context` is the ANSI-stripped visible pane frame;
+sessionName, kind: 'permission'|'question'|'idle', createdAt, toolName?,
+toolSummary?, message?, cwd?, context?, options?: {n, label}[],
+acknowledgedAt? }`. `context` is the ANSI-stripped visible pane frame;
`options` is present only when the dialog's numbered choices parsed
confidently; `acknowledgedAt` marks an item a human has already looked at
(see `/viewed` below) and tells clients not to re-arm its tab alert. Listing
@@ -466,7 +474,7 @@ Design: [`approvals-inbox-plan.md`](approvals-inbox-plan.md).
first, `422 OPERATION_FAILED` when the session refused input.
- `POST /api/v1/approvals/:id/dismiss` removes the item without keystrokes.
- `POST /api/v1/approvals/session/:sessionId/viewed` → `{ sessionId,
- acknowledged: itemId | null }`. Marks the session's pending **idle** item as
+acknowledged: itemId | null }`. Marks the session's pending **idle** item as
seen by a human (the web UI calls it when you open the session's tab): the
item stays pending and answerable, but stops arming the yellow tab alert on
every client, including after a reload. Permission/question items are never
@@ -479,6 +487,47 @@ re-captured, or the item acknowledged), `approval:resolved` (`{ id, sessionId, k
`resolution` one of `answered | resolved_in_terminal | superseded |
session_ended | dismissed | expired`).
+## Reboot restore
+
+A host reboot takes the tmux server down with it, so every pane dies and the
+board comes up empty. At boot Codeman works out which sessions the reboot
+destroyed and holds that plan in memory, and these endpoints let a client offer
+it to the user. Nothing creates a pane until the user asks: the boot-time reboot
+heuristic decides whether to ASK, never whether to act.
+
+Claude-mode sessions only (others carry their conversation id in their own
+config object); remote and docker sessions are never offered, because both need
+another host or container to be up. The plan is in-memory, so a server restart
+drops it and the offer is gone — the conversations themselves are unaffected,
+since they live in the CLI's own transcript store and stay reachable from the
+Resume list. A plan nobody spends expires after 24 hours.
+
+- `GET /api/v1/reboot-restore` → `{ sessions: RestorableSession[],
+scrollbackRestored: false }`, ownership-scoped in multi-user mode.
+ `RestorableSession`: `{ id, name?, workingDir, mode, owner? }`. The persisted
+ record itself is never sent. `scrollbackRestored` is always `false` and exists
+ so a client states it: a restored session is a NEW pane, so the conversation
+ continues and the terminal history does not.
+- `POST /api/v1/reboot-restore/restore` with `{ sessionIds?: string[] }` (omit
+ to restore everything the caller can see) → `{ restored: RestorableSession[],
+skipped: { sessionId, reason }[] }`. `reason` is one of `workspace-missing`
+ (the directory is gone), `workspace-forbidden` (it is outside the caller's
+ workspace in multi-user mode), `already-live` (the conversation is already
+ open, typically resumed by hand from the Resume list), `capacity-reached`
+ (the global or per-user session cap), or `rebuild-failed` (the agent would not
+ start, most often a CLI binary missing from the server's PATH).
+ `409 CONFLICT` when that caller already has a restore running. Entries are
+ removed from the plan before any pane is built, so a double-click cannot put
+ two panes on one conversation; anything that never became a pane goes back on
+ offer, except `already-live`, which cannot stop being true. A restored session
+ comes back attached, idle and disarmed — respawn controllers and Ralph loops
+ are never re-armed automatically.
+- `POST /api/v1/reboot-restore/dismiss` → `{ dismissed: n }`. Drops the offer
+ for everything the caller can see.
+
+Each rebuilt session also emits the ordinary `session:created` SSE event, so
+clients other than the one that clicked pick it up without refetching.
+
## Read My Mind intent profiles
Per-case profiles of what the user is trying to accomplish: user/agent-stated
@@ -491,7 +540,7 @@ user guide: [`readmymind.md`](readmymind.md).
- `GET /api/v1/sessions/:id/intent` -> `{ intent: IntentProfile }` for the
session's case. `IntentProfile`: `{ key, workingDir, updatedAt, goals,
- recentPrompts: { ts, sessionId, text }[] }` (prompts oldest first, FIFO cap
+recentPrompts: { ts, sessionId, text }[] }` (prompts oldest first, FIFO cap
50, each <= 500 chars). A case with nothing recorded answers an empty
profile with `updatedAt: 0`; nothing is persisted by reads.
- `PUT /api/v1/sessions/:id/intent` with `{ goals }` (<= 8192 chars, strict
@@ -524,7 +573,7 @@ same speech-to-text service the CLI's own `/voice` mode uses. Gated on the synce
[`claude-voice-plan.md`](claude-voice-plan.md).
- `GET /api/v1/voice/status` -> `{ available, reason?, subscriptionType?,
- expiresAt? }`. `reason` is `disabled` (setting off), `no-credentials` (nobody
+expiresAt? }`. `reason` is `disabled` (setting off), `no-credentials` (nobody
signed in to Claude Code on the server), `expired` (the access token elapsed;
running any Claude session refreshes it) or `malformed`. The OAuth token
itself is never returned by this or any other endpoint.
diff --git a/src/reboot-restore.ts b/src/reboot-restore.ts
index c8b6bd5e..e226138c 100644
--- a/src/reboot-restore.ts
+++ b/src/reboot-restore.ts
@@ -57,6 +57,12 @@ export interface RebootEvidence {
*
* This heuristic decides whether to ASK, never whether to act. A wrong yes costs
* the user a banner they dismiss, because the restore itself waits for a click.
+ *
+ * ⚠️ `os.uptime()` reports the HOST's uptime, which a container shares. A Codeman
+ * running in Docker therefore sees a long uptime after its own container restarts,
+ * the boot test fails, and no banner appears. The feature is effectively off for
+ * containerized installs. That is the safe direction to fail in, and fixing it
+ * needs a boot signal the container actually owns rather than a wider heuristic.
*/
export function looksLikeHostReboot(evidence: RebootEvidence): boolean {
if (evidence.deadSessionCount === 0) return false;
@@ -80,7 +86,14 @@ export function resolveResumeConversationId(state: SessionState): string {
return chainTail || state.resumeSessionId || state.id;
}
-/** Why one session was passed over. Reported for logging and assertions. */
+/**
+ * Why one session was passed over. Reported for logging and shown to the user.
+ *
+ * The first six are decided before anything is built. `capacity-reached` and
+ * `rebuild-failed` can only happen once a click is spending the plan, and they
+ * are the two the banner must not confuse with a missing workspace: one means
+ * "try again after closing something", the other means the CLI would not start.
+ */
export interface RebootRestoreRejection {
sessionId: string;
reason:
@@ -91,7 +104,10 @@ export interface RebootRestoreRejection {
| 'unsupported-mode'
| 'no-working-dir'
| 'workspace-missing'
- | 'already-live';
+ | 'workspace-forbidden'
+ | 'already-live'
+ | 'capacity-reached'
+ | 'rebuild-failed';
}
/** One restorable session, as the banner shows it and the rebuild replays it. */
diff --git a/src/session-env-clamp.ts b/src/session-env-clamp.ts
index edaecd5c..b9abf3a6 100644
--- a/src/session-env-clamp.ts
+++ b/src/session-env-clamp.ts
@@ -3,10 +3,16 @@
*
* A session's `envOverrides` can hand back privilege that the per-CLI config
* clamp removed, so a non-granted owner's overrides get the privileged keys
- * stripped before the session is built. Two callers need that today. The create
- * and resume routes clamp what a request asked for, and the reboot-restore route
- * clamps what a persisted record carried, because a record written while its
- * owner held a grant must not replay that grant after the grant is gone.
+ * stripped before the session is built. The create and resume routes are what
+ * this bites on: they clamp what a request asked for.
+ *
+ * The reboot-restore route calls it as defence in depth, and today it can strip
+ * nothing. `Session.getEnvOverridesForPersist()` keeps only `CLAUDE_CODE_*` and
+ * `CLAUDE_CONFIG_DIR` out of a session's overrides, claude's `privilegedEnvKeys`
+ * are the five `ANTHROPIC_*` names, and that pass admits claude alone — so a
+ * persisted record cannot carry a clamped key. The call is there for the day the
+ * persisted set widens. The grant re-resolution that does bite on that path is
+ * `resolveClaudeModeForUsername`, which recomputes the permission mode.
*
* This lives outside `web/routes` on purpose. The question it answers is about
* session privilege rather than about HTTP, and `cron/cron-service.ts` sets the
diff --git a/src/web/ports/session-port.ts b/src/web/ports/session-port.ts
index 61e02be3..83918763 100644
--- a/src/web/ports/session-port.ts
+++ b/src/web/ports/session-port.ts
@@ -4,6 +4,7 @@
*/
import type { Session } from '../../session.js';
+import type { SessionState } from '../../types.js';
export interface SessionPort {
readonly sessions: ReadonlyMap;
@@ -12,5 +13,16 @@ export interface SessionPort {
setupSessionListeners(session: Session): Promise;
persistSessionState(session: Session): void;
persistSessionStateNow(session: Session): void;
+ /**
+ * Re-apply the persisted state a freshly CONSTRUCTED session does not carry:
+ * the pin, token and cost totals, auto-compact, auto-clear, auto-resume, nice
+ * priority, the flicker filter and the custom-model selection.
+ *
+ * A `Session` built from a record holds only what its constructor takes, so
+ * persisting it would otherwise REPLACE the fuller record with the reduced one.
+ * Call this before the first persist, and before `startInteractive()`, because
+ * the custom-model selection has to reach the pane's environment.
+ */
+ reapplyPersistedSessionState(session: Session, saved: SessionState): Promise;
getSessionStateWithRespawn(session: Session): unknown;
}
diff --git a/src/web/public/app.js b/src/web/public/app.js
index 1f0ae409..7c670497 100644
--- a/src/web/public/app.js
+++ b/src/web/public/app.js
@@ -957,7 +957,9 @@ class CodemanApp {
this.registerServiceWorker();
// Fetch tunnel status for header indicator (desktop only)
this.loadTunnelStatus();
- // Ask whether a host reboot left sessions worth rebuilding (banner, never automatic)
+ // Ask whether a host reboot left sessions worth rebuilding (banner, never
+ // automatic). handleInit() re-reads it on every SSE init; this covers the
+ // path where that event never arrives.
this.initRebootRestoreBanner?.();
// Share a single settings fetch between both consumers
const settingsPromise = fetch('/api/settings').then(r => r.ok ? r.json() : null).then(env => env?.data ?? null).catch(() => null);
@@ -3761,6 +3763,12 @@ class CodemanApp {
// a fresh load / reconnect (authoritative; wins over the localStorage restore).
if (data.planUsage) this.updatePlanUsageChip(data.planUsage);
+ // A board left open across a host reboot reconnects HERE, to a server that came
+ // back with an empty session list. The reboot-restore offer is built at boot,
+ // before any client could be listening, so re-read it on every init rather than
+ // only on the page-load path.
+ this.refreshRebootRestoreBanner?.();
+
// Update version displays (header and toolbar)
if (data.version) {
const versionEl = this.$('versionDisplay');
diff --git a/src/web/public/mobile.css b/src/web/public/mobile.css
index 9be29153..d324a94f 100644
--- a/src/web/public/mobile.css
+++ b/src/web/public/mobile.css
@@ -3240,6 +3240,41 @@ html:is([data-skin="paper-gray"], [data-skin="solarized-light"], [data-skin="cat
already reserves that space), so it needs the same safe-area padding as the
other banners. The overlay is fixed and handles its own insets.
============================================================================ */
+@media (max-width: 599px) {
+ /* Reboot-restore banner: the same treatment as the offline banner below. Its
+ text and note are nowrap and the two buttons cannot shrink, so without this
+ the actions are pushed off a phone-width viewport and become unreachable. */
+ .reboot-restore-banner {
+ padding: 0.4rem 0.5rem;
+ padding-left: calc(0.5rem + var(--safe-area-left));
+ padding-right: calc(0.5rem + var(--safe-area-right));
+ font-size: 0.7rem;
+ gap: 0.4rem;
+ }
+
+ /* The session names and the scrollback note are the first things to go. The
+ count plus the two buttons carry the message on their own, and the note
+ survives as the accept button's title. */
+ .reboot-restore-banner-detail,
+ .reboot-restore-banner-note {
+ display: none;
+ }
+
+ .reboot-restore-banner-text {
+ overflow: hidden;
+ text-overflow: ellipsis;
+ }
+
+ .reboot-restore-banner-accept,
+ .reboot-restore-banner-dismiss {
+ padding: 0.25rem 0.5rem;
+ }
+
+ .reboot-restore-banner-accept {
+ margin-left: auto;
+ }
+}
+
@media (max-width: 599px) {
.offline-banner {
padding: 0.4rem 0.5rem;
diff --git a/src/web/public/reboot-restore-ui.js b/src/web/public/reboot-restore-ui.js
index 0869eebe..aec1de81 100644
--- a/src/web/public/reboot-restore-ui.js
+++ b/src/web/public/reboot-restore-ui.js
@@ -8,7 +8,9 @@
* reboot guess is a heuristic and a wrong automatic restore would spawn CLI
* processes nobody asked for.
*
- * Seeded once from `GET /api/reboot-restore` on init. Restore posts to
+ * Seeded from `GET /api/reboot-restore` on init and again on every SSE reconnect,
+ * because the tab most likely to want this is one that was open across the reboot
+ * and reconnects to a server that came back up with an empty board. Restore posts to
* `POST /api/reboot-restore/restore`, Dismiss posts to
* `POST /api/reboot-restore/dismiss`, and either way the banner goes away. The
* restored sessions arrive as ordinary `session:created` events, so no extra
@@ -25,6 +27,24 @@
* @loadorder 11.7 of 17, after approvals-ui.js
*/
+/** Plain-language wording for one skip reason, for the toast after a restore. */
+function rebootSkipReason(reason) {
+ switch (reason) {
+ case 'workspace-missing':
+ return 'workspace is gone';
+ case 'workspace-forbidden':
+ return 'workspace is outside your space';
+ case 'already-live':
+ return 'already open';
+ case 'capacity-reached':
+ return 'session limit reached';
+ case 'rebuild-failed':
+ return 'the agent would not start';
+ default:
+ return reason;
+ }
+}
+
Object.assign(CodemanApp.prototype, {
/** Ask the server whether a reboot left anything on offer, and show the banner if so. */
async initRebootRestoreBanner() {
@@ -59,6 +79,9 @@ Object.assign(CodemanApp.prototype, {
detail.textContent = count > 4 ? `${names}, …` : names;
detail.title = sessions.map((s) => `${s.name || s.id}\n${s.workingDir}`).join('\n\n');
}
+ const accept = this.$('rebootRestoreBannerAccept');
+ // The note is hidden at phone width, so the warning travels on the button too.
+ if (accept) accept.title = 'Conversations return; terminal history does not.';
banner.hidden = false;
},
@@ -66,8 +89,9 @@ Object.assign(CodemanApp.prototype, {
async restoreRebootSessions() {
const button = this.$('rebootRestoreBannerAccept');
if (button) button.disabled = true;
- const res = await this._apiPost('/api/reboot-restore/restore', {});
- const body = res && res.ok ? await res.json().catch(() => null) : null;
+ // _apiJson unwraps the { success, data } envelope every /api response carries;
+ // reading the outer object would report every count as zero.
+ const body = await this._apiJson('/api/reboot-restore/restore', { method: 'POST', body: {} });
if (!body) {
if (button) button.disabled = false;
this.showToast?.('Could not restore the sessions', 'error');
@@ -82,10 +106,21 @@ Object.assign(CodemanApp.prototype, {
this.showToast?.(`Restored ${restored} ${noun}. Terminal history did not survive the reboot.`, 'success');
}
if (skipped > 0) {
- this.showToast?.(`${skipped} could not be restored (workspace gone, or already open)`, 'warning');
+ // Each reason means a different next step for the user, so they are not
+ // collapsed into one message: capacity clears by closing something, a
+ // failed start usually means the CLI is not on the server's PATH.
+ const reasons = new Set((body.skipped ?? []).map((s) => s.reason));
+ this.showToast?.(`${skipped} not restored: ${[...reasons].map(rebootSkipReason).join('; ')}`, 'warning');
}
},
+ /** Re-read the offer after a reconnect, for a tab that was open across the reboot. */
+ async refreshRebootRestoreBanner() {
+ const data = await this._apiJson('/api/reboot-restore');
+ this._rebootRestoreSessions = data?.sessions ?? [];
+ this.renderRebootRestoreBanner();
+ },
+
/** Drop the offer. The Resume list still reaches every one of these conversations. */
async dismissRebootRestore() {
this._rebootRestoreSessions = [];
diff --git a/src/web/reboot-restore-registry.ts b/src/web/reboot-restore-registry.ts
index e265fa62..789217eb 100644
--- a/src/web/reboot-restore-registry.ts
+++ b/src/web/reboot-restore-registry.ts
@@ -24,8 +24,9 @@
* - Spending is take-then-build: `take()` removes entries synchronously, before
* the route's first `await`, so a double-click or two devices cannot both
* reach the same entry and put two panes on one conversation.
- * - One restore runs at a time. `beginSpending()` single-flights the route, so
- * two concurrent clicks cannot interleave pane creation.
+ * - One restore runs at a time per owner. `beginSpending()` single-flights the
+ * route, so two concurrent clicks cannot interleave pane creation for the same
+ * user, while two different users never block each other.
*
* @dependencies reboot-restore (RebootRestoreEntry)
* @consumedby web/server (plan build at boot), web/routes/reboot-restore-routes
@@ -46,8 +47,13 @@ export class RebootRestoreRegistry {
private entries = new Map();
/** When the boot pass built the plan, in ms since the epoch. */
private builtAt = 0;
- /** True while a restore route call is between its take and its last pane. */
- private spending = false;
+ /**
+ * Owners with a restore in flight, between its take and its last pane.
+ * Keyed by owner so one user's restore does not turn another user's click into
+ * a conflict; `take()` already guarantees no two callers get the same entry.
+ * Single-user mode has one key, `undefined`, so it behaves as one global flight.
+ */
+ private spending = new Set();
/** Replace the plan with what the boot pass found. An empty list clears it. */
set(entries: readonly RebootRestoreEntry[]): void {
@@ -92,9 +98,11 @@ export class RebootRestoreRegistry {
/**
* Put entries back after a rebuild never got as far as creating a pane.
*
- * Used for the click-time rejections, so a conversation the user resumed by
- * hand meanwhile does not silently vanish from the banner while a workspace
- * that came back stays offered.
+ * Used for the click-time rejections that may resolve themselves: a workspace
+ * that comes back, a capacity limit the user makes room under, a CLI that
+ * starts once its binary is on the PATH. A conversation the user resumed by
+ * hand is NOT put back, because that one cannot stop being true, and an entry
+ * the banner keeps re-offering forever is noise only Dismiss can clear.
*/
restore(entries: readonly RebootRestoreEntry[]): void {
for (const entry of entries) this.entries.set(entry.sessionId, entry);
@@ -110,24 +118,25 @@ export class RebootRestoreRegistry {
}
/**
- * Claim the right to run a restore, or report that one is already running.
- * Callers that get `true` must call `endSpending()` in a `finally`.
+ * Claim the right to run a restore for one owner, or report that owner already
+ * has one running. Callers that get `true` must call `endSpending()` in a
+ * `finally` with the same owner.
*/
- beginSpending(): boolean {
- if (this.spending) return false;
- this.spending = true;
+ beginSpending(owner?: string): boolean {
+ if (this.spending.has(owner)) return false;
+ this.spending.add(owner);
return true;
}
- endSpending(): void {
- this.spending = false;
+ endSpending(owner?: string): void {
+ this.spending.delete(owner);
}
/** Test hook: forget everything, including the single-flight claim. */
reset(): void {
this.entries.clear();
this.builtAt = 0;
- this.spending = false;
+ this.spending.clear();
}
private dropIfExpired(): void {
diff --git a/src/web/routes/reboot-restore-routes.ts b/src/web/routes/reboot-restore-routes.ts
index 580b17c6..a03c74af 100644
--- a/src/web/routes/reboot-restore-routes.ts
+++ b/src/web/routes/reboot-restore-routes.ts
@@ -27,7 +27,14 @@ import { FastifyInstance } from 'fastify';
import { existsSync } from 'node:fs';
import { ApiErrorCode, createErrorResponse, getErrorMessage } from '../../types.js';
import { RebootRestoreRequestSchema } from '../schemas.js';
-import { parseBody, getAuthUser, canAccessOwned } from '../route-helpers.js';
+import {
+ parseBody,
+ getAuthUser,
+ canAccessOwned,
+ ownerFor,
+ isWorkingDirAllowed,
+ sessionCapacityMessage,
+} from '../route-helpers.js';
import { rebootRestoreRegistry } from '../reboot-restore-registry.js';
import { rejectAlreadyLive, type RebootRestoreEntry, type RebootRestoreRejection } from '../../reboot-restore.js';
import { clampEnvOverridesForOwner } from '../../session-env-clamp.js';
@@ -76,38 +83,65 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
app.post('/api/reboot-restore/restore', async (req, reply) => {
const body = parseBody(RebootRestoreRequestSchema, req.body, 'Invalid reboot restore request');
+ const user = getAuthUser(req);
const canAccess = accessorFor(req);
+ const owner = ownerFor(req);
// Take BEFORE the first await: a second click must find nothing to spend.
- if (!rebootRestoreRegistry.beginSpending()) {
+ // The flight is per owner, because `take()` already guarantees two callers
+ // never receive the same entry, so one user's restore need not block another's.
+ if (!rebootRestoreRegistry.beginSpending(owner)) {
return reply.code(409).send(createErrorResponse(ApiErrorCode.CONFLICT, 'A reboot restore is already running'));
}
const taken = rebootRestoreRegistry.take(canAccess, body.sessionIds);
+ // Entries nothing built a pane for, returned to the plan on every exit path
+ // including a throw. Without this a failure between here and the loop would
+ // spend the offer and rebuild nothing, and the plan cannot be rebuilt.
+ const unspent = new Set(taken);
try {
if (taken.length === 0) return { restored: [], skipped: [] };
// The plan was built at boot and the board has moved on since. A conversation
// the user resumed by hand from the Resume list is already on screen, and a
- // second pane on it would fight the first for the same transcript.
+ // second pane on it would fight the first for the same transcript. This one
+ // is never re-offered: unlike a missing workspace, it cannot stop being true.
const liveSessionIds = new Set(ctx.sessions.keys());
const liveConversationIds = new Set(
[...ctx.sessions.values()].map((session) => session.claudeSessionId).filter((id): id is string => !!id)
);
const { restore, skipped } = rejectAlreadyLive(taken, liveSessionIds, liveConversationIds);
- // An entry nothing rebuilt stays on offer rather than disappearing silently.
- rebootRestoreRegistry.restore(skipped.map((s) => taken.find((e) => e.sessionId === s.sessionId)!));
+ for (const entry of taken) {
+ if (skipped.some((s) => s.sessionId === entry.sessionId)) unspent.delete(entry);
+ }
const restored: ReturnType[] = [];
const failures: RebootRestoreRejection[] = [...skipped];
const workspaceHooksEnabled = await ctx.getWorkspaceHooksEnabled();
for (const entry of restore) {
+ // Capacity is re-checked per iteration, because this loop is itself
+ // creating the sessions it counts. The offer can be a day old, so the
+ // board may be fuller now than the plan assumed.
+ const capMsg = sessionCapacityMessage(ctx.sessions, entry.owner);
+ if (capMsg) {
+ failures.push({ sessionId: entry.sessionId, reason: 'capacity-reached' });
+ continue;
+ }
// A repo can be deleted between the boot that planned this and the click.
if (!existsSync(entry.workingDir)) {
failures.push({ sessionId: entry.sessionId, reason: 'workspace-missing' });
continue;
}
+ // Multi-user workspace separation: the create route confines a non-admin's
+ // workingDir to their own case space, and a grant can be withdrawn between
+ // the session's creation and this restore, so the confinement is re-run
+ // rather than inherited from the record.
+ if (!isWorkingDirAllowed(user, entry.workingDir)) {
+ failures.push({ sessionId: entry.sessionId, reason: 'workspace-forbidden' });
+ unspent.delete(entry);
+ continue;
+ }
try {
const saved = entry.state;
const claudeModeConfig = await ctx.getClaudeModeConfig();
@@ -145,9 +179,14 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
});
await ctx.addSession(session);
- ctx.persistSessionState(session);
await ctx.setupSessionListeners(session);
+ // Before the pane spawns: the custom-model selection reaches it through
+ // the environment. Before the first persist: a constructed session holds
+ // none of this, so persisting it first would replace the fuller record
+ // with the reduced one and drop the pin that keeps it from being pruned.
+ await ctx.reapplyPersistedSessionState(session, saved);
await session.startInteractive();
+ ctx.persistSessionState(session);
// A session without its workspace hooks goes silently blind: no stop or
// idle events for respawn, no Approvals Inbox item, no red tab on a
@@ -166,10 +205,22 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
ctx.broadcast(SseEvent.SessionCreated, ctx.getSessionStateWithRespawn(session));
restored.push(toBannerItem(entry));
} catch (err) {
- // One workspace that has gone missing must not stop the rest of the pass.
+ // One entry that will not start must not stop the rest of the pass, and
+ // must not leave a registered session with no pane behind it: by this
+ // point the session is in `ctx.sessions`, holds a tab-layout slot and has
+ // listeners, and the commonest cause is a CLI binary that is not on the
+ // PATH of a freshly booted machine.
console.error(`[reboot-restore] failed to rebuild ${entry.sessionId}:`, err);
- failures.push({ sessionId: entry.sessionId, reason: 'workspace-missing' });
+ await ctx
+ .cleanupSession(entry.sessionId, true, 'reboot restore failed to start the session')
+ .catch((cleanupErr: unknown) =>
+ console.error(`[reboot-restore] cleanup after a failed rebuild failed: ${getErrorMessage(cleanupErr)}`)
+ );
+ failures.push({ sessionId: entry.sessionId, reason: 'rebuild-failed' });
+ // Left on offer: the user can put the binary back and click again.
+ continue;
}
+ unspent.delete(entry);
}
if (restored.length > 0) {
@@ -181,7 +232,10 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
return { restored, skipped: failures };
} finally {
- rebootRestoreRegistry.endSpending();
+ // Anything that never became a pane goes back on offer, including after a
+ // throw, so a transient failure costs a retry rather than the whole plan.
+ rebootRestoreRegistry.restore([...unspent]);
+ rebootRestoreRegistry.endSpending(owner);
}
});
diff --git a/src/web/server.ts b/src/web/server.ts
index 065172d5..a7339328 100644
--- a/src/web/server.ts
+++ b/src/web/server.ts
@@ -671,6 +671,7 @@ export class WebServer extends EventEmitter {
setupSessionListeners: this.setupSessionListeners.bind(this),
persistSessionState: this.persistSessionState.bind(this),
persistSessionStateNow: this._persistSessionStateNow.bind(this),
+ reapplyPersistedSessionState: this.reapplyPersistedSessionState.bind(this),
getSessionStateWithRespawn: this.getSessionStateWithRespawn.bind(this),
// EventPort
broadcast: this.broadcast.bind(this),
@@ -2907,6 +2908,53 @@ export class WebServer extends EventEmitter {
return restore.length;
}
+ /**
+ * Re-apply the persisted state that a `Session` constructor does not take.
+ *
+ * The reboot-restore route builds a session from a record rather than
+ * attaching to a surviving pane, so everything the constructor has no
+ * parameter for starts at its default. Persisting such a session writes
+ * `toState()` wholesale, which would REPLACE the record with the reduced
+ * version — and for a pinned session that is worse than losing a setting,
+ * because `cleanupSessionsByIds()` keeps a record only while it is pinned, so
+ * dropping the pin hands the record to the next stale sweep.
+ *
+ * Respawn and Ralph are deliberately NOT re-armed here: a machine that just
+ * came up is the worst moment to turn an autonomous run loose, and the user
+ * re-arms what they want.
+ */
+ async reapplyPersistedSessionState(session: Session, saved: SessionState): Promise {
+ // The custom-model env has to be rebuilt from the endpoint store: the persist
+ // deliberately keeps the injected VALUES out of state.json, so only the
+ // bookkeeping survives a restart and the values are re-derived here.
+ const savedCustomModel = (saved as { __customModel?: CustomModelBookkeeping }).__customModel;
+ if (savedCustomModel) {
+ session.setCustomModel(savedCustomModel, await this._rebuildCustomModelEnv(session, savedCustomModel));
+ }
+ if (saved.pinned) session.setPinned(true);
+ if (saved.autoCompactEnabled !== undefined || saved.autoCompactThreshold !== undefined) {
+ session.setAutoCompact(saved.autoCompactEnabled ?? false, saved.autoCompactThreshold, saved.autoCompactPrompt);
+ }
+ if (saved.autoClearEnabled !== undefined || saved.autoClearThreshold !== undefined) {
+ session.setAutoClear(saved.autoClearEnabled ?? false, saved.autoClearThreshold);
+ }
+ if (saved.autoResumeEnabled) {
+ session.restoreAutoResume(true, saved.autoResumeAt);
+ }
+ if (saved.inputTokens !== undefined || saved.outputTokens !== undefined || saved.totalCost !== undefined) {
+ session.restoreTokens(saved.inputTokens ?? 0, saved.outputTokens ?? 0, saved.totalCost ?? 0);
+ // Seed the daily-usage baseline, or the restored totals are counted again as new usage.
+ this.lastRecordedTokens.set(session.id, {
+ input: saved.inputTokens ?? 0,
+ output: saved.outputTokens ?? 0,
+ });
+ }
+ if (saved.niceEnabled !== undefined || saved.niceValue !== undefined) {
+ session.setNice({ enabled: saved.niceEnabled, niceValue: saved.niceValue });
+ }
+ if (saved.flickerFilterEnabled !== undefined) session.flickerFilterEnabled = saved.flickerFilterEnabled;
+ }
+
private async restoreMuxSessions(): Promise {
try {
// Reconcile mux sessions to find which ones are still alive (also discovers unknown ones)
diff --git a/test/mocks/mock-route-context.ts b/test/mocks/mock-route-context.ts
index 8a1bd884..401538be 100644
--- a/test/mocks/mock-route-context.ts
+++ b/test/mocks/mock-route-context.ts
@@ -61,6 +61,7 @@ export function createMockRouteContext(options?: {
setupSessionListeners: vi.fn(async () => {}),
persistSessionState: vi.fn(),
persistSessionStateNow: vi.fn(),
+ reapplyPersistedSessionState: vi.fn(async () => {}),
getSessionStateWithRespawn: vi.fn((s: MockSession) => s.toState()),
// -- EventPort --
@@ -149,6 +150,7 @@ export function createMockRouteContext(options?: {
clearRespawnConfig: vi.fn(),
updateRespawnConfig: vi.fn(),
setHistoryLimit: vi.fn(async () => {}),
+ startStatsCollection: vi.fn(),
},
runSummaryTrackers: new Map(),
activePlanOrchestrators: new Map(),
diff --git a/test/routes/reboot-restore-rebuild-failure.test.ts b/test/routes/reboot-restore-rebuild-failure.test.ts
new file mode 100644
index 00000000..f8f58f58
--- /dev/null
+++ b/test/routes/reboot-restore-rebuild-failure.test.ts
@@ -0,0 +1,210 @@
+/**
+ * Reboot-restore route: what happens when a rebuild gets part-way and then fails.
+ *
+ * The other route test file deliberately uses workspaces that do not exist, so it
+ * never reaches `new Session()`. This one mocks the `Session` module so the route
+ * runs its whole construction path — `addSession`, `setupSessionListeners`,
+ * `reapplyPersistedSessionState`, `startInteractive` — and then throws where a
+ * real one would when the CLI binary is missing from a freshly booted machine's
+ * PATH. Without the mock there is no way to exercise that path, which is how the
+ * original version of this route shipped a session leak the tests could not see.
+ *
+ * It also covers the session caps, because those too are only reachable once the
+ * route is actually willing to build something.
+ */
+import { describe, it, expect, afterEach, vi, beforeEach } from 'vitest';
+import Fastify, { type FastifyInstance } from 'fastify';
+import fastifyCookie from '@fastify/cookie';
+
+/** Set per test: whether the mocked `startInteractive()` rejects. */
+let startShouldThrow = false;
+
+vi.mock('../../src/session.js', () => ({
+ Session: class {
+ id: string;
+ mode: string;
+ name?: string;
+ workingDir: string;
+ owner?: string;
+ claudeSessionId: string | null = null;
+ constructor(config: { id: string; mode?: string; name?: string; workingDir: string; owner?: string }) {
+ this.id = config.id;
+ this.mode = config.mode ?? 'claude';
+ this.name = config.name;
+ this.workingDir = config.workingDir;
+ this.owner = config.owner;
+ }
+ async startInteractive() {
+ if (startShouldThrow) throw new Error('spawn claude ENOENT');
+ }
+ /** The mock route context projects a session through this on broadcast. */
+ toState() {
+ return { id: this.id, mode: this.mode, name: this.name, workingDir: this.workingDir, owner: this.owner };
+ }
+ },
+}));
+
+const { registerRebootRestoreRoutes } = await import('../../src/web/routes/reboot-restore-routes.js');
+const { rebootRestoreRegistry } = await import('../../src/web/reboot-restore-registry.js');
+const { installRouteErrorHandler } = await import('../../src/web/route-error-handler.js');
+const { httpStatusForErrorCode } = await import('../../src/types.js');
+const { createMockRouteContext } = await import('../mocks/index.js');
+type ApiErrorCode = import('../../src/types.js').ApiErrorCode;
+type RebootRestoreEntry = import('../../src/reboot-restore.js').RebootRestoreEntry;
+type SessionState = import('../../src/types.js').SessionState;
+
+/** A real directory, so the route's workspace checks pass and it reaches the build. */
+const WORKSPACE = process.cwd();
+
+function offerEntry(sessionId: string, owner?: string): RebootRestoreEntry {
+ return {
+ sessionId,
+ name: `session ${sessionId}`,
+ workingDir: WORKSPACE,
+ owner,
+ mode: 'claude',
+ resumeConversationId: `conv-${sessionId}`,
+ state: {
+ id: sessionId,
+ pid: null,
+ status: 'idle',
+ workingDir: WORKSPACE,
+ currentTaskId: null,
+ createdAt: 1_760_000_000_000,
+ mode: 'claude',
+ owner,
+ } as SessionState,
+ };
+}
+
+async function createHarness(ctx: ReturnType): Promise {
+ const app = Fastify({ logger: false });
+ await app.register(fastifyCookie);
+ registerRebootRestoreRoutes(app, ctx as never);
+ app.addHook('preSerialization', (req, reply, payload: unknown, done) => {
+ if (!req.url.startsWith('/api')) return done(null, payload);
+ if (payload === null || typeof payload !== 'object') return done(null, payload);
+ const p = payload as { success?: unknown; errorCode?: unknown };
+ if (p.success === false) {
+ if (reply.statusCode === 200 && typeof p.errorCode === 'string') {
+ reply.code(httpStatusForErrorCode(p.errorCode as ApiErrorCode));
+ }
+ return done(null, payload);
+ }
+ if (p.success === true) return done(null, payload);
+ return done(null, { success: true, data: payload });
+ });
+ installRouteErrorHandler(app);
+ await app.ready();
+ return app;
+}
+
+beforeEach(() => {
+ startShouldThrow = false;
+});
+
+afterEach(() => {
+ rebootRestoreRegistry.reset();
+ vi.clearAllMocks();
+});
+
+describe('a rebuild that fails after the session is registered', () => {
+ it('reports why it failed rather than blaming the workspace', async () => {
+ startShouldThrow = true;
+ rebootRestoreRegistry.set([offerEntry('a')]);
+ const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
+ const app = await createHarness(ctx);
+
+ const res = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data;
+ expect(res.restored).toEqual([]);
+ // Not `workspace-missing`: the directory is there, the agent would not start.
+ expect(res.skipped).toEqual([{ sessionId: 'a', reason: 'rebuild-failed' }]);
+ await app.close();
+ });
+
+ it('does not leave a registered session with no pane behind it', async () => {
+ startShouldThrow = true;
+ rebootRestoreRegistry.set([offerEntry('a')]);
+ const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
+ const app = await createHarness(ctx);
+
+ await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} });
+ // The session reached ctx.sessions via addSession; the route has to take it
+ // back out, or the board shows a tab whose pane never existed.
+ expect(ctx.cleanupSession).toHaveBeenCalledWith('a', true, expect.any(String));
+ await app.close();
+ });
+
+ it('keeps the entry on offer, so the user can fix the PATH and click again', async () => {
+ startShouldThrow = true;
+ rebootRestoreRegistry.set([offerEntry('a')]);
+ const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
+ const app = await createHarness(ctx);
+
+ await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} });
+ const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data;
+ expect(left.sessions.map((s: { id: string }) => s.id)).toEqual(['a']);
+ await app.close();
+ });
+});
+
+describe('a rebuild that succeeds', () => {
+ it('re-applies the persisted state before the record is written again', async () => {
+ rebootRestoreRegistry.set([offerEntry('a')]);
+ const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
+ const app = await createHarness(ctx);
+
+ const res = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data;
+ expect(res.restored.map((s: { id: string }) => s.id)).toEqual(['a']);
+ // A session built from a record carries none of the pin, token totals or
+ // custom-model selection, so persisting it first would replace the fuller
+ // record with the reduced one.
+ expect(ctx.reapplyPersistedSessionState).toHaveBeenCalled();
+ const reapplyOrder = (ctx.reapplyPersistedSessionState as ReturnType).mock.invocationCallOrder[0];
+ const persistOrder = (ctx.persistSessionState as ReturnType).mock.invocationCallOrder[0];
+ expect(reapplyOrder).toBeLessThan(persistOrder);
+ await app.close();
+ });
+
+ it('tells every other board about the rebuilt session', async () => {
+ rebootRestoreRegistry.set([offerEntry('a')]);
+ const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
+ const app = await createHarness(ctx);
+
+ await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} });
+ expect(ctx.broadcast).toHaveBeenCalledWith('session:created', expect.anything());
+ await app.close();
+ });
+
+ it('spends the entry, so it is no longer on offer', async () => {
+ rebootRestoreRegistry.set([offerEntry('a')]);
+ const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
+ const app = await createHarness(ctx);
+
+ await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} });
+ const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data;
+ expect(left.sessions).toEqual([]);
+ await app.close();
+ });
+});
+
+describe('the session caps', () => {
+ it('stops restoring at the global cap and leaves the rest on offer', async () => {
+ rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]);
+ const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
+ // Fill the board to the documented maximum of 50 concurrent sessions.
+ for (let i = 0; i < 50; i += 1) {
+ ctx.sessions.set(`filler-${i}`, { id: `filler-${i}`, owner: undefined } as never);
+ }
+ const app = await createHarness(ctx);
+
+ const res = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data;
+ expect(res.restored).toEqual([]);
+ expect(res.skipped.map((s: { reason: string }) => s.reason)).toEqual(['capacity-reached', 'capacity-reached']);
+
+ // Refused rather than lost: closing a session and clicking again works.
+ const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data;
+ expect(left.sessions.map((s: { id: string }) => s.id).sort()).toEqual(['a', 'b']);
+ await app.close();
+ });
+});
diff --git a/test/routes/reboot-restore-routes.test.ts b/test/routes/reboot-restore-routes.test.ts
index c16cddac..dcb7d7ed 100644
--- a/test/routes/reboot-restore-routes.test.ts
+++ b/test/routes/reboot-restore-routes.test.ts
@@ -23,6 +23,13 @@ import type { RebootRestoreEntry } from '../../src/reboot-restore.js';
import type { SessionState } from '../../src/types.js';
async function createHarness(authUser?: { username: string; role: 'admin' | 'user' }): Promise {
+ return createHarnessWithCtx(createMockRouteContext(), authUser);
+}
+
+async function createHarnessWithCtx(
+ ctx: ReturnType,
+ authUser?: { username: string; role: 'admin' | 'user' }
+): Promise {
const app = Fastify({ logger: false });
await app.register(fastifyCookie);
if (authUser) {
@@ -30,7 +37,7 @@ async function createHarness(authUser?: { username: string; role: 'admin' | 'use
(req as unknown as { authUser: typeof authUser }).authUser = authUser;
});
}
- registerRebootRestoreRoutes(app, createMockRouteContext() as never);
+ registerRebootRestoreRoutes(app, ctx as never);
app.addHook('preSerialization', (req, reply, payload: unknown, done) => {
if (!req.url.startsWith('/api')) return done(null, payload);
@@ -106,18 +113,36 @@ describe('GET /api/reboot-restore', () => {
});
describe('POST /api/reboot-restore/restore', () => {
- it('spends the offer, so a second click finds nothing left to spend', async () => {
+ it('reports a workspace that is gone, and keeps offering it in case it comes back', async () => {
rebootRestoreRegistry.set([offerEntry('a')]);
const app = await createHarness();
const first = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data;
- // The workspace is gone, so nothing was rebuilt — but the entry was taken.
expect(first.restored).toEqual([]);
expect(first.skipped).toEqual([{ sessionId: 'a', reason: 'workspace-missing' }]);
- const second = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data;
- expect(second.restored).toEqual([]);
- expect(second.skipped).toEqual([]);
+ // Nothing was built, so the entry goes back: a repo can be restored from a
+ // backup between two clicks, and losing the offer would be unrecoverable.
+ const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data;
+ expect(left.sessions.map((s: { id: string }) => s.id)).toEqual(['a']);
+ await app.close();
+ });
+
+ it('never re-offers a conversation that is already open', async () => {
+ const entry = offerEntry('a');
+ rebootRestoreRegistry.set([entry]);
+ const app = await createHarness();
+ const ctx = createMockRouteContext({ sessionId: entry.sessionId });
+ // A session with that id is live, which is what the Resume list would produce.
+ const liveApp = await createHarnessWithCtx(ctx);
+
+ const res = (await liveApp.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data;
+ expect(res.skipped).toEqual([{ sessionId: 'a', reason: 'already-live' }]);
+
+ // Unlike a missing workspace, this one is dropped: it cannot stop being true.
+ const left = (await liveApp.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data;
+ expect(left.sessions).toEqual([]);
+ await liveApp.close();
await app.close();
});
@@ -132,8 +157,9 @@ describe('POST /api/reboot-restore/restore', () => {
});
expect(res.json().data.skipped).toEqual([{ sessionId: 'b', reason: 'workspace-missing' }]);
+ // 'a' was never taken, and 'b' came back because no pane was built for it.
const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data;
- expect(left.sessions.map((s: { id: string }) => s.id)).toEqual(['a']);
+ expect(left.sessions.map((s: { id: string }) => s.id).sort()).toEqual(['a', 'b']);
await app.close();
});
@@ -150,12 +176,13 @@ describe('POST /api/reboot-restore/restore', () => {
it('turns a second concurrent restore away rather than interleaving it', async () => {
rebootRestoreRegistry.set([offerEntry('a')]);
- // Claimed by a restore already in flight.
- expect(rebootRestoreRegistry.beginSpending()).toBe(true);
+ // Claimed by a restore already in flight for this same owner (undefined in
+ // single-user mode, which is what the harness runs as).
+ expect(rebootRestoreRegistry.beginSpending(undefined)).toBe(true);
const app = await createHarness();
const res = await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} });
expect(res.statusCode).toBe(409);
- rebootRestoreRegistry.endSpending();
+ rebootRestoreRegistry.endSpending(undefined);
await app.close();
});
});
From fa52753e8bb0a33996c8fd774c101b917a7acf94 Mon Sep 17 00:00:00 2001
From: Michael Grundberg
Date: Wed, 16 Sep 2026 15:04:35 +0200
Subject: [PATCH 12/28] fix(sessions): undo a failed rebuild without deleting
the user's data
A second review of the previous commit found that its own repair for the
session leak introduced three defects, all from reaching for
cleanupSession() to undo a half-built session. That function is the
user-initiated delete, not an undo.
It banked the session's historical token and cost totals into the lifetime
figures, and a reboot never runs cleanup, so those totals had never been
counted before; every failed rebuild added them again. It saw the pin that
had just been restored and demoted the record to `stopped`, which this pass
reads as the durable marker of a deliberate kill, so a pinned session whose
rebuild failed became permanently unrestorable. And it recursively removed
`.claude-images` from the working directory, which belongs to the workspace
rather than to the session, so a failed rebuild destroyed the pasted images
of any other live session in that repo.
discardPartiallyBuiltSession() now undoes only what the construction did:
the map entry, the tab-layout slot, the listeners and any pane the launch
created before throwing. The persisted record, the lifetime totals, the
Ralph state and the workspace's files are left alone.
Re-applying the persisted state also splits in two, which removes the first
two defects at the root rather than only at the call site. The half that
shapes the pane, the custom-model environment and the nice priority, still
runs before the spawn. The half that is the session's own history now runs
after it, so a session whose pane never started carries no totals and no pin
for anything downstream to misread.
The rest of that review. The multi-user workspace confinement re-check read
the requesting user's grant, and returns true for an admin, so the case its
own comment described was the one it missed; it now resolves the entry
owner's grant through isWorkingDirAllowedForUsername, the way cron does. A
forbidden workspace goes back on offer, matching both the registry's stated
contract and the API reference. The client re-reads the plan after a restore
instead of blanking the banner, so entries the server put back stay
reachable, and a 409 now says a restore is already running rather than
reporting a failure. A dismiss arriving mid-restore wins, through a
generation counter the route carries across its take. The re-application
also restores the tab colour, the image-watcher flag and the original
pinnedAt, via a new Session.restorePin that does not re-stamp the pin time.
The phone breakpoint gains min-width: 0, without which a nowrap flex item
never shrinks and the buttons still overflow, and it folds into the existing
phone block.
Ralph's loop configuration still does not survive a restore, because
toState() reads it off a live tracker and there is no way to keep it without
arming the loop. The method now says so rather than leaving it implied.
Tests. The capacity test could not fail on the property it existed for: it
filled the board past the cap before the loop, so a single pre-loop check
would have passed it. It now leaves one seat, so only a per-iteration check
restores exactly one entry. New tests cover the ordering around the spawn,
a throw before the loop returning the whole plan and releasing the flight,
the dismiss-during-restore race, and that the failure path calls the narrow
discard rather than the delete. The shared mock context gains the port
method it was missing, which is what made the first run of these tests fail
for the wrong reason.
Refs #411
Co-Authored-By: Claude Opus 5 (1M context)
---
src/session.ts | 13 +++
src/web/ports/session-port.ts | 24 +++--
src/web/public/mobile.css | 6 +-
src/web/public/reboot-restore-ui.js | 22 +++--
src/web/reboot-restore-registry.ts | 21 +++-
src/web/routes/reboot-restore-routes.ts | 43 +++++---
src/web/server.ts | 91 ++++++++++++++---
test/mocks/mock-route-context.ts | 3 +
.../reboot-restore-rebuild-failure.test.ts | 97 ++++++++++++++++++-
9 files changed, 273 insertions(+), 47 deletions(-)
diff --git a/src/session.ts b/src/session.ts
index 21d5ac89..16720e59 100644
--- a/src/session.ts
+++ b/src/session.ts
@@ -1499,6 +1499,19 @@ export class Session extends EventEmitter {
this._pinnedAt = pinned ? Date.now() : null;
}
+ /**
+ * Restore a pin from a persisted record, keeping the moment it was pinned.
+ *
+ * `setPinned()` stamps `pinnedAt` with now, which is right for a user pinning a
+ * session and wrong for a restore: the session-manager orders its pinned group
+ * by that stamp, so a restored session would jump to the front of a list it had
+ * been sitting further down.
+ */
+ restorePin(pinned: boolean, pinnedAt?: number): void {
+ this._pinned = pinned;
+ this._pinnedAt = pinned ? (pinnedAt ?? Date.now()) : null;
+ }
+
get flickerFilterEnabled(): boolean {
return this._flickerFilterEnabled;
}
diff --git a/src/web/ports/session-port.ts b/src/web/ports/session-port.ts
index 83918763..85add1d8 100644
--- a/src/web/ports/session-port.ts
+++ b/src/web/ports/session-port.ts
@@ -14,15 +14,27 @@ export interface SessionPort {
persistSessionState(session: Session): void;
persistSessionStateNow(session: Session): void;
/**
- * Re-apply the persisted state a freshly CONSTRUCTED session does not carry:
- * the pin, token and cost totals, auto-compact, auto-clear, auto-resume, nice
- * priority, the flicker filter and the custom-model selection.
+ * Re-apply the persisted state a freshly CONSTRUCTED session does not carry.
*
* A `Session` built from a record holds only what its constructor takes, so
* persisting it would otherwise REPLACE the fuller record with the reduced one.
- * Call this before the first persist, and before `startInteractive()`, because
- * the custom-model selection has to reach the pane's environment.
+ * Two phases: `before-spawn` shapes the pane (the custom-model environment and
+ * the nice priority) and must precede `startInteractive()`; `after-spawn` is
+ * the session's own history (the pin, token and cost totals, auto-compact,
+ * auto-clear, auto-resume, colour, image watcher, flicker filter) and must NOT
+ * land on a session whose pane failed to start.
*/
- reapplyPersistedSessionState(session: Session, saved: SessionState): Promise;
+ reapplyPersistedSessionState(
+ session: Session,
+ saved: SessionState,
+ phase: 'before-spawn' | 'after-spawn'
+ ): Promise;
+ /**
+ * Undo a session that was registered but never got a working pane: the map
+ * entry, its tab-layout slot, and any pane the launch created before throwing.
+ * Unlike {@link cleanupSession} it leaves the persisted record, the lifetime
+ * token totals, the Ralph state and the workspace's own files untouched.
+ */
+ discardPartiallyBuiltSession(sessionId: string): Promise;
getSessionStateWithRespawn(session: Session): unknown;
}
diff --git a/src/web/public/mobile.css b/src/web/public/mobile.css
index d324a94f..f364c701 100644
--- a/src/web/public/mobile.css
+++ b/src/web/public/mobile.css
@@ -3260,7 +3260,11 @@ html:is([data-skin="paper-gray"], [data-skin="solarized-light"], [data-skin="cat
display: none;
}
+ /* A flex item will not shrink below its content width at the default
+ `min-width: auto`, so without this the nowrap text pushes the buttons off a
+ 360px viewport and the ellipsis never engages. */
.reboot-restore-banner-text {
+ min-width: 0;
overflow: hidden;
text-overflow: ellipsis;
}
@@ -3273,9 +3277,7 @@ html:is([data-skin="paper-gray"], [data-skin="solarized-light"], [data-skin="cat
.reboot-restore-banner-accept {
margin-left: auto;
}
-}
-@media (max-width: 599px) {
.offline-banner {
padding: 0.4rem 0.5rem;
padding-left: calc(0.5rem + var(--safe-area-left));
diff --git a/src/web/public/reboot-restore-ui.js b/src/web/public/reboot-restore-ui.js
index aec1de81..1cdc8da8 100644
--- a/src/web/public/reboot-restore-ui.js
+++ b/src/web/public/reboot-restore-ui.js
@@ -23,7 +23,7 @@
*
* @mixin Extends CodemanApp.prototype via Object.assign
* @dependency app.js (CodemanApp class, showToast)
- * @dependency api-client.js at runtime (this._apiJson / this._apiPost)
+ * @dependency api-client.js at runtime (this._api / this._apiJson)
* @loadorder 11.7 of 17, after approvals-ui.js
*/
@@ -89,9 +89,15 @@ Object.assign(CodemanApp.prototype, {
async restoreRebootSessions() {
const button = this.$('rebootRestoreBannerAccept');
if (button) button.disabled = true;
- // _apiJson unwraps the { success, data } envelope every /api response carries;
- // reading the outer object would report every count as zero.
- const body = await this._apiJson('/api/reboot-restore/restore', { method: 'POST', body: {} });
+ const res = await this._api('/api/reboot-restore/restore', { method: 'POST', body: {} });
+ if (res && res.status === 409) {
+ if (button) button.disabled = false;
+ this.showToast?.('A restore is already running', 'info');
+ return;
+ }
+ // The uniform envelope wraps every /api payload; reading the outer object
+ // would report every count as zero.
+ const body = res && res.ok ? (await res.json().catch(() => null))?.data : null;
if (!body) {
if (button) button.disabled = false;
this.showToast?.('Could not restore the sessions', 'error');
@@ -99,8 +105,12 @@ Object.assign(CodemanApp.prototype, {
}
const restored = body.restored?.length ?? 0;
const skipped = body.skipped?.length ?? 0;
- this._rebootRestoreSessions = [];
- this.renderRebootRestoreBanner();
+ // Re-read rather than clearing: the server puts back anything it could not
+ // build for a reason that may pass, such as a session limit or an agent that
+ // would not start, and blanking the banner here would put those entries out
+ // of reach until a reload.
+ await this.refreshRebootRestoreBanner();
+ if (button) button.disabled = false;
if (restored > 0) {
const noun = restored === 1 ? 'conversation' : 'conversations';
this.showToast?.(`Restored ${restored} ${noun}. Terminal history did not survive the reboot.`, 'success');
diff --git a/src/web/reboot-restore-registry.ts b/src/web/reboot-restore-registry.ts
index 789217eb..28b4a1d7 100644
--- a/src/web/reboot-restore-registry.ts
+++ b/src/web/reboot-restore-registry.ts
@@ -47,6 +47,13 @@ export class RebootRestoreRegistry {
private entries = new Map();
/** When the boot pass built the plan, in ms since the epoch. */
private builtAt = 0;
+ /**
+ * Bumped by anything that invalidates entries a restore is already holding.
+ * A Dismiss arriving mid-restore must win: without this the route's `finally`
+ * would put its unspent entries back and resurrect the offer the user just
+ * cleared, with a fresh 24-hour life.
+ */
+ private generation = 0;
/**
* Owners with a restore in flight, between its take and its last pane.
* Keyed by owner so one user's restore does not turn another user's click into
@@ -59,6 +66,12 @@ export class RebootRestoreRegistry {
set(entries: readonly RebootRestoreEntry[]): void {
this.entries = new Map(entries.map((entry) => [entry.sessionId, entry]));
this.builtAt = entries.length > 0 ? Date.now() : 0;
+ this.generation += 1;
+ }
+
+ /** The current generation, for a caller that will later return entries. */
+ currentGeneration(): number {
+ return this.generation;
}
/**
@@ -104,7 +117,10 @@ export class RebootRestoreRegistry {
* hand is NOT put back, because that one cannot stop being true, and an entry
* the banner keeps re-offering forever is noise only Dismiss can clear.
*/
- restore(entries: readonly RebootRestoreEntry[]): void {
+ restore(entries: readonly RebootRestoreEntry[], generation?: number): void {
+ // A dismiss (or a fresh boot plan) since the caller took these entries means
+ // they are no longer wanted back.
+ if (generation !== undefined && generation !== this.generation) return;
for (const entry of entries) this.entries.set(entry.sessionId, entry);
if (entries.length > 0 && this.builtAt === 0) this.builtAt = Date.now();
}
@@ -114,6 +130,8 @@ export class RebootRestoreRegistry {
const removable = [...this.entries.values()].filter((entry) => canAccess(entry.owner));
for (const entry of removable) this.entries.delete(entry.sessionId);
if (this.entries.size === 0) this.builtAt = 0;
+ // Any restore currently in flight must not put its entries back afterwards.
+ this.generation += 1;
return removable.length;
}
@@ -137,6 +155,7 @@ export class RebootRestoreRegistry {
this.entries.clear();
this.builtAt = 0;
this.spending.clear();
+ this.generation += 1;
}
private dropIfExpired(): void {
diff --git a/src/web/routes/reboot-restore-routes.ts b/src/web/routes/reboot-restore-routes.ts
index a03c74af..636f2cd5 100644
--- a/src/web/routes/reboot-restore-routes.ts
+++ b/src/web/routes/reboot-restore-routes.ts
@@ -32,7 +32,7 @@ import {
getAuthUser,
canAccessOwned,
ownerFor,
- isWorkingDirAllowed,
+ isWorkingDirAllowedForUsername,
sessionCapacityMessage,
} from '../route-helpers.js';
import { rebootRestoreRegistry } from '../reboot-restore-registry.js';
@@ -83,7 +83,6 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
app.post('/api/reboot-restore/restore', async (req, reply) => {
const body = parseBody(RebootRestoreRequestSchema, req.body, 'Invalid reboot restore request');
- const user = getAuthUser(req);
const canAccess = accessorFor(req);
const owner = ownerFor(req);
@@ -93,6 +92,7 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
if (!rebootRestoreRegistry.beginSpending(owner)) {
return reply.code(409).send(createErrorResponse(ApiErrorCode.CONFLICT, 'A reboot restore is already running'));
}
+ const generation = rebootRestoreRegistry.currentGeneration();
const taken = rebootRestoreRegistry.take(canAccess, body.sessionIds);
// Entries nothing built a pane for, returned to the plan on every exit path
// including a throw. Without this a failure between here and the loop would
@@ -136,10 +136,15 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
// Multi-user workspace separation: the create route confines a non-admin's
// workingDir to their own case space, and a grant can be withdrawn between
// the session's creation and this restore, so the confinement is re-run
- // rather than inherited from the record.
- if (!isWorkingDirAllowed(user, entry.workingDir)) {
+ // rather than inherited from the record. Keyed on the OWNER, not on the
+ // caller: an admin spending another user's entry must be held to that
+ // user's confinement, and `isWorkingDirAllowed` would wave an admin
+ // through. The same reason the two grant re-checks below read
+ // `saved.owner`.
+ if (!(await isWorkingDirAllowedForUsername(entry.owner, entry.workingDir))) {
+ // Left on offer: a withdrawn grant can be restored, unlike an already-open
+ // conversation, so this is not the permanent kind of refusal.
failures.push({ sessionId: entry.sessionId, reason: 'workspace-forbidden' });
- unspent.delete(entry);
continue;
}
try {
@@ -180,12 +185,16 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
await ctx.addSession(session);
await ctx.setupSessionListeners(session);
- // Before the pane spawns: the custom-model selection reaches it through
- // the environment. Before the first persist: a constructed session holds
- // none of this, so persisting it first would replace the fuller record
- // with the reduced one and drop the pin that keeps it from being pruned.
- await ctx.reapplyPersistedSessionState(session, saved);
+ // Shapes the pane, so it has to land before the CLI process starts.
+ await ctx.reapplyPersistedSessionState(session, saved, 'before-spawn');
await session.startInteractive();
+ // The session's own history, applied only once the pane exists: on a
+ // failed start these totals would belong to a session that never ran.
+ // Both halves precede the first persist, because a constructed session
+ // carries none of this and `toState()` is written wholesale, so
+ // persisting first would replace the fuller record with the reduced one
+ // and drop the pin that keeps it from being pruned.
+ await ctx.reapplyPersistedSessionState(session, saved, 'after-spawn');
ctx.persistSessionState(session);
// A session without its workspace hooks goes silently blind: no stop or
@@ -211,10 +220,15 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
// listeners, and the commonest cause is a CLI binary that is not on the
// PATH of a freshly booted machine.
console.error(`[reboot-restore] failed to rebuild ${entry.sessionId}:`, err);
+ // Not cleanupSession(): that is the user-initiated delete, and it would
+ // count this session's historical tokens into the lifetime totals, demote
+ // a pinned record to `stopped` (which this pass reads as an intentional
+ // kill, making the session permanently unrestorable) and delete the
+ // workspace's `.claude-images`. This undoes only the construction.
await ctx
- .cleanupSession(entry.sessionId, true, 'reboot restore failed to start the session')
- .catch((cleanupErr: unknown) =>
- console.error(`[reboot-restore] cleanup after a failed rebuild failed: ${getErrorMessage(cleanupErr)}`)
+ .discardPartiallyBuiltSession(entry.sessionId)
+ .catch((discardErr: unknown) =>
+ console.error(`[reboot-restore] discarding a failed rebuild failed: ${getErrorMessage(discardErr)}`)
);
failures.push({ sessionId: entry.sessionId, reason: 'rebuild-failed' });
// Left on offer: the user can put the binary back and click again.
@@ -234,7 +248,8 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
} finally {
// Anything that never became a pane goes back on offer, including after a
// throw, so a transient failure costs a retry rather than the whole plan.
- rebootRestoreRegistry.restore([...unspent]);
+ // Passing the generation makes a Dismiss that landed mid-restore win.
+ rebootRestoreRegistry.restore([...unspent], generation);
rebootRestoreRegistry.endSpending(owner);
}
});
diff --git a/src/web/server.ts b/src/web/server.ts
index a7339328..474d8cd2 100644
--- a/src/web/server.ts
+++ b/src/web/server.ts
@@ -672,6 +672,7 @@ export class WebServer extends EventEmitter {
persistSessionState: this.persistSessionState.bind(this),
persistSessionStateNow: this._persistSessionStateNow.bind(this),
reapplyPersistedSessionState: this.reapplyPersistedSessionState.bind(this),
+ discardPartiallyBuiltSession: this.discardPartiallyBuiltSession.bind(this),
getSessionStateWithRespawn: this.getSessionStateWithRespawn.bind(this),
// EventPort
broadcast: this.broadcast.bind(this),
@@ -2919,19 +2920,42 @@ export class WebServer extends EventEmitter {
* because `cleanupSessionsByIds()` keeps a record only while it is pinned, so
* dropping the pin hands the record to the next stale sweep.
*
- * Respawn and Ralph are deliberately NOT re-armed here: a machine that just
- * came up is the worst moment to turn an autonomous run loose, and the user
- * re-arms what they want.
+ * Split in two phases because the two halves have opposite timing needs:
+ *
+ * - `before-spawn` shapes the pane itself, so it has to land before the CLI
+ * process starts. The custom-model selection is an environment injection and
+ * the nice priority is applied to the spawn.
+ * - `after-spawn` is the session's own accumulated history. It must NOT land
+ * on a session whose pane failed to start: the totals would then belong to a
+ * session that never ran, and any later cleanup would add them to the
+ * lifetime figures a second time.
+ *
+ * Respawn and Ralph are deliberately NOT re-armed: a machine that just came up
+ * is the worst moment to turn an autonomous run loose, and the user re-arms
+ * what they want. Ralph's loop CONFIGURATION does not survive either, because
+ * `toState()` reads `ralphEnabled` and the completion phrase off a live
+ * tracker, and there is no way to hold them without arming the loop.
*/
- async reapplyPersistedSessionState(session: Session, saved: SessionState): Promise {
- // The custom-model env has to be rebuilt from the endpoint store: the persist
- // deliberately keeps the injected VALUES out of state.json, so only the
- // bookkeeping survives a restart and the values are re-derived here.
- const savedCustomModel = (saved as { __customModel?: CustomModelBookkeeping }).__customModel;
- if (savedCustomModel) {
- session.setCustomModel(savedCustomModel, await this._rebuildCustomModelEnv(session, savedCustomModel));
+ async reapplyPersistedSessionState(
+ session: Session,
+ saved: SessionState,
+ phase: 'before-spawn' | 'after-spawn'
+ ): Promise {
+ if (phase === 'before-spawn') {
+ // The custom-model env has to be rebuilt from the endpoint store: the persist
+ // deliberately keeps the injected VALUES out of state.json, so only the
+ // bookkeeping survives a restart and the values are re-derived here.
+ const savedCustomModel = (saved as { __customModel?: CustomModelBookkeeping }).__customModel;
+ if (savedCustomModel) {
+ session.setCustomModel(savedCustomModel, await this._rebuildCustomModelEnv(session, savedCustomModel));
+ }
+ if (saved.niceEnabled !== undefined || saved.niceValue !== undefined) {
+ session.setNice({ enabled: saved.niceEnabled, niceValue: saved.niceValue });
+ }
+ return;
}
- if (saved.pinned) session.setPinned(true);
+
+ if (saved.pinned) session.restorePin(true, saved.pinnedAt);
if (saved.autoCompactEnabled !== undefined || saved.autoCompactThreshold !== undefined) {
session.setAutoCompact(saved.autoCompactEnabled ?? false, saved.autoCompactThreshold, saved.autoCompactPrompt);
}
@@ -2949,12 +2973,51 @@ export class WebServer extends EventEmitter {
output: saved.outputTokens ?? 0,
});
}
- if (saved.niceEnabled !== undefined || saved.niceValue !== undefined) {
- session.setNice({ enabled: saved.niceEnabled, niceValue: saved.niceValue });
- }
+ if (saved.color) session.setColor(saved.color);
+ if (saved.imageWatcherEnabled !== undefined) session.imageWatcherEnabled = saved.imageWatcherEnabled;
if (saved.flickerFilterEnabled !== undefined) session.flickerFilterEnabled = saved.flickerFilterEnabled;
}
+ /**
+ * Undo a session that was registered but never got a working pane.
+ *
+ * Deliberately NOT `cleanupSession()`, which is the user-initiated delete: that
+ * path adds the session's token totals to the lifetime figures, demotes a
+ * pinned record to `stopped` (the durable marker of an intentional kill, which
+ * would make the session permanently ineligible for a reboot restore), drops
+ * the persisted Ralph state, and recursively removes `.claude-images` from the
+ * WORKING DIRECTORY, which belongs to the workspace rather than to this session
+ * and may hold another live session's pasted images.
+ *
+ * This undoes only what the failed construction did: the map entry, the tab
+ * layout slot `registerSessionWithLayout()` took, and any pane the CLI launch
+ * managed to create before it threw. The persisted record is left exactly as it
+ * was, so the session stays restorable on the next attempt.
+ */
+ async discardPartiallyBuiltSession(sessionId: string): Promise {
+ const session = this.sessions.get(sessionId);
+ if (!session) return;
+ this.sessions.delete(sessionId);
+ this.sse.cleanupSessionBatches(sessionId);
+ this.persistDeb.cancelKey(sessionId);
+ try {
+ session.removeAllListeners();
+ await session.stop?.();
+ } catch (err) {
+ console.warn(`[Server] stopping a partially built session failed: ${getErrorMessage(err)}`);
+ }
+ try {
+ await this.mux.killSession(sessionId);
+ } catch {
+ // The pane may never have been created; nothing to kill is the normal case.
+ }
+ try {
+ await this.tabLayouts.sessionsRemoved([{ id: sessionId, owner: session.owner }]);
+ } catch (err) {
+ console.warn(`[Server] releasing the tab layout slot failed: ${getErrorMessage(err)}`);
+ }
+ }
+
private async restoreMuxSessions(): Promise {
try {
// Reconcile mux sessions to find which ones are still alive (also discovers unknown ones)
diff --git a/test/mocks/mock-route-context.ts b/test/mocks/mock-route-context.ts
index 401538be..9cb26d6a 100644
--- a/test/mocks/mock-route-context.ts
+++ b/test/mocks/mock-route-context.ts
@@ -62,6 +62,9 @@ export function createMockRouteContext(options?: {
persistSessionState: vi.fn(),
persistSessionStateNow: vi.fn(),
reapplyPersistedSessionState: vi.fn(async () => {}),
+ discardPartiallyBuiltSession: vi.fn(async (id: string) => {
+ sessions.delete(id);
+ }),
getSessionStateWithRespawn: vi.fn((s: MockSession) => s.toState()),
// -- EventPort --
diff --git a/test/routes/reboot-restore-rebuild-failure.test.ts b/test/routes/reboot-restore-rebuild-failure.test.ts
index f8f58f58..f1a35417 100644
--- a/test/routes/reboot-restore-rebuild-failure.test.ts
+++ b/test/routes/reboot-restore-rebuild-failure.test.ts
@@ -18,6 +18,8 @@ import fastifyCookie from '@fastify/cookie';
/** Set per test: whether the mocked `startInteractive()` rejects. */
let startShouldThrow = false;
+/** Ordering log, so a test can assert what ran before the pane spawned. */
+const callOrder: string[] = [];
vi.mock('../../src/session.js', () => ({
Session: class {
@@ -35,6 +37,7 @@ vi.mock('../../src/session.js', () => ({
this.owner = config.owner;
}
async startInteractive() {
+ callOrder.push('startInteractive');
if (startShouldThrow) throw new Error('spawn claude ENOENT');
}
/** The mock route context projects a session through this on broadcast. */
@@ -101,6 +104,7 @@ async function createHarness(ctx: ReturnType): Pr
beforeEach(() => {
startShouldThrow = false;
+ callOrder.length = 0;
});
afterEach(() => {
@@ -131,7 +135,12 @@ describe('a rebuild that fails after the session is registered', () => {
await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} });
// The session reached ctx.sessions via addSession; the route has to take it
// back out, or the board shows a tab whose pane never existed.
- expect(ctx.cleanupSession).toHaveBeenCalledWith('a', true, expect.any(String));
+ expect(ctx.discardPartiallyBuiltSession).toHaveBeenCalledWith('a');
+ expect(ctx.sessions.has('a')).toBe(false);
+ // NOT the user-initiated delete: that would bank this session's historical
+ // tokens into the lifetime totals, demote a pinned record to `stopped`, and
+ // delete the workspace's .claude-images.
+ expect(ctx.cleanupSession).not.toHaveBeenCalled();
await app.close();
});
@@ -166,6 +175,23 @@ describe('a rebuild that succeeds', () => {
await app.close();
});
+ it('shapes the pane before it spawns, and restores the history after', async () => {
+ rebootRestoreRegistry.set([offerEntry('a')]);
+ const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
+ (ctx.reapplyPersistedSessionState as ReturnType).mockImplementation(
+ async (_s: unknown, _saved: unknown, phase: string) => {
+ callOrder.push(`reapply:${phase}`);
+ }
+ );
+ const app = await createHarness(ctx);
+
+ await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} });
+ // The custom-model environment has to reach the process; the token totals
+ // must not land on a session whose pane never started.
+ expect(callOrder).toEqual(['reapply:before-spawn', 'startInteractive', 'reapply:after-spawn']);
+ await app.close();
+ });
+
it('tells every other board about the rebuilt session', async () => {
rebootRestoreRegistry.set([offerEntry('a')]);
const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
@@ -189,10 +215,30 @@ describe('a rebuild that succeeds', () => {
});
describe('the session caps', () => {
- it('stops restoring at the global cap and leaves the rest on offer', async () => {
+ it('counts the sessions it is itself creating, not just the ones it started with', async () => {
+ rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]);
+ const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
+ // One seat short of the documented maximum of 50, counting the session the
+ // mock context seeds. A check that ran once before the loop would restore
+ // BOTH entries; only a per-iteration check refuses the second.
+ for (let i = 0; i < 48; i += 1) {
+ ctx.sessions.set(`filler-${i}`, { id: `filler-${i}`, owner: undefined } as never);
+ }
+ const app = await createHarness(ctx);
+
+ const res = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data;
+ expect(res.restored.map((s: { id: string }) => s.id)).toEqual(['a']);
+ expect(res.skipped).toEqual([{ sessionId: 'b', reason: 'capacity-reached' }]);
+
+ // Refused rather than lost: closing a session and clicking again works.
+ const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data;
+ expect(left.sessions.map((s: { id: string }) => s.id)).toEqual(['b']);
+ await app.close();
+ });
+
+ it('refuses every entry when the board is already at the cap', async () => {
rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]);
const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
- // Fill the board to the documented maximum of 50 concurrent sessions.
for (let i = 0; i < 50; i += 1) {
ctx.sessions.set(`filler-${i}`, { id: `filler-${i}`, owner: undefined } as never);
}
@@ -201,10 +247,53 @@ describe('the session caps', () => {
const res = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data;
expect(res.restored).toEqual([]);
expect(res.skipped.map((s: { reason: string }) => s.reason)).toEqual(['capacity-reached', 'capacity-reached']);
+ await app.close();
+ });
+});
- // Refused rather than lost: closing a session and clicking again works.
+describe('a failure before any entry is considered', () => {
+ it('returns the whole plan rather than spending it', async () => {
+ rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]);
+ const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
+ (ctx.getWorkspaceHooksEnabled as ReturnType).mockRejectedValue(new Error('settings unreadable'));
+ const app = await createHarness(ctx);
+
+ const res = await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} });
+ expect(res.statusCode).toBeGreaterThanOrEqual(500);
+
+ // The plan cannot be rebuilt once boot has pruned the records, so a throw
+ // anywhere in the route has to hand the entries back.
const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data;
expect(left.sessions.map((s: { id: string }) => s.id).sort()).toEqual(['a', 'b']);
await app.close();
});
+
+ it('releases the single flight, so the next click is not refused', async () => {
+ rebootRestoreRegistry.set([offerEntry('a')]);
+ const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
+ (ctx.getWorkspaceHooksEnabled as ReturnType).mockRejectedValue(new Error('settings unreadable'));
+ const app = await createHarness(ctx);
+
+ await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} });
+ expect(rebootRestoreRegistry.beginSpending(undefined)).toBe(true);
+ rebootRestoreRegistry.endSpending(undefined);
+ await app.close();
+ });
+});
+
+describe('a dismiss that lands while a restore is running', () => {
+ it('wins, rather than being undone when the restore hands its entries back', async () => {
+ const entries = [offerEntry('a')];
+ rebootRestoreRegistry.set(entries);
+ const generation = rebootRestoreRegistry.currentGeneration();
+ const taken = rebootRestoreRegistry.take(() => true);
+ expect(taken).toHaveLength(1);
+
+ // The user clears the banner while the restore is still working.
+ rebootRestoreRegistry.clear(() => true);
+ // The restore finishes and tries to put its unspent entry back.
+ rebootRestoreRegistry.restore(taken, generation);
+
+ expect(rebootRestoreRegistry.list(() => true)).toEqual([]);
+ });
});
From 71ed7b127c0ce4f14c1f1a71a9abc3ee21fe1c50 Mon Sep 17 00:00:00 2001
From: Michael Grundberg
Date: Wed, 16 Sep 2026 15:34:47 +0200
Subject: [PATCH 13/28] fix(sessions): make the discard a real inverse of the
construction
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Third review of the reboot-restore branch. The narrow discard the previous
commit introduced avoided everything cleanupSession() did wrongly, and in
dropping so much of it also dropped four things it had to keep.
The worst broke the retry the whole design rests on. setupSessionListeners()
returns early while sessionListenerRefs still holds the session id, and the
discard never cleared that entry. So the advertised flow — a rebuild fails
because the agent binary is missing, the user fixes their PATH and clicks
again — reused the same id, wired no listeners at all, and produced a tab
that never showed output, never updated its status and never persisted. That
is worse than the leak the discard was added to prevent. Three more
registrations leaked with it: a RunSummaryTracker and its interval, an image
watcher on the workspace, and the Ralph fix-plan watcher. The discard now
undoes each registration setupSessionListeners() makes, in its order, and
the per-session custom-model config directory, which holds the endpoint's
API key literally and which nothing else would ever remove.
The image-watcher flag was restored after the code that reads it, so a
session came back reporting the feature as on with nothing watching. It
moves to the before-spawn phase, and that phase now runs before the
listeners rather than after them.
The generation counter that lets a mid-restore dismiss win was global while
clear() is ownership-scoped, so one user's dismiss discarded another user's
unspent entries, permanently, because nothing rebuilds an in-memory plan. It
is now per owner. Bumping only the owners of entries the dismiss removed was
not enough either: take() has already emptied the plan by then, so a dismiss
landing mid-restore saw nothing of that owner's to remove and invalidated
nothing. The owners that matter are those with a restore in flight, filtered
by what the dismissing user may access, and that is what clear() now bumps.
Plan expiry bumps too, so a restore straddling the 24-hour boundary cannot
hand entries back and give an expired plan another full day.
Tests. discardPartiallyBuiltSession had no test at all: the only
implementation any test ran was the mock's one-line stub, which is why every
defect above was invisible. test/discard-partially-built-session.ts drives
the real WebServer, and the retry assertion fails if the listener refs are
left behind — verified by reverting the fix. The dismiss-race test drove the
registry by hand, so deleting the route's generation argument left it green;
it now goes through the route, and two further tests cover the multi-user
cases.
The mock context has now gone stale twice, because route tests pass it as
`ctx as never` and tsconfig.json includes only src, so nothing ever compares
it to the ports. A type-level guard is therefore inert — I wrote one and
confirmed it never fires. test/mocks/mock-route-context-completeness.ts
compares the mock's keys against WebServer.createRouteContext() at runtime
instead, and names what is missing.
Also: the API reference now says workspace-forbidden is judged against the
owner's grant, the banner's module header no longer claims Restore always
dismisses it, and the detail span gets the same min-width: 0 the phone rule
already needed.
Refs #411
Co-Authored-By: Claude Opus 5 (1M context)
---
docs/api-reference.md | 5 +-
src/web/public/reboot-restore-ui.js | 10 +-
src/web/public/styles.css | 4 +
src/web/reboot-restore-registry.ts | 73 ++++++++---
src/web/routes/reboot-restore-routes.ts | 23 ++--
src/web/server.ts | 58 +++++++--
test/discard-partially-built-session.test.ts | 113 ++++++++++++++++++
.../mock-route-context-completeness.test.ts | 28 +++++
.../reboot-restore-rebuild-failure.test.ts | 64 ++++++++--
9 files changed, 322 insertions(+), 56 deletions(-)
create mode 100644 test/discard-partially-built-session.test.ts
create mode 100644 test/mocks/mock-route-context-completeness.test.ts
diff --git a/docs/api-reference.md b/docs/api-reference.md
index 85349ee3..4bca9768 100644
--- a/docs/api-reference.md
+++ b/docs/api-reference.md
@@ -511,8 +511,9 @@ scrollbackRestored: false }`, ownership-scoped in multi-user mode.
- `POST /api/v1/reboot-restore/restore` with `{ sessionIds?: string[] }` (omit
to restore everything the caller can see) → `{ restored: RestorableSession[],
skipped: { sessionId, reason }[] }`. `reason` is one of `workspace-missing`
- (the directory is gone), `workspace-forbidden` (it is outside the caller's
- workspace in multi-user mode), `already-live` (the conversation is already
+ (the directory is gone), `workspace-forbidden` (in multi-user mode it is
+ outside the workspace of the user the session belongs to, re-checked against
+ that owner's current grant rather than the caller's), `already-live` (the conversation is already
open, typically resumed by hand from the Resume list), `capacity-reached`
(the global or per-user session cap), or `rebuild-failed` (the agent would not
start, most often a CLI binary missing from the server's PATH).
diff --git a/src/web/public/reboot-restore-ui.js b/src/web/public/reboot-restore-ui.js
index 1cdc8da8..2a3599dd 100644
--- a/src/web/public/reboot-restore-ui.js
+++ b/src/web/public/reboot-restore-ui.js
@@ -11,10 +11,12 @@
* Seeded from `GET /api/reboot-restore` on init and again on every SSE reconnect,
* because the tab most likely to want this is one that was open across the reboot
* and reconnects to a server that came back up with an empty board. Restore posts to
- * `POST /api/reboot-restore/restore`, Dismiss posts to
- * `POST /api/reboot-restore/dismiss`, and either way the banner goes away. The
- * restored sessions arrive as ordinary `session:created` events, so no extra
- * rendering is needed here.
+ * `POST /api/reboot-restore/restore` and Dismiss posts to
+ * `POST /api/reboot-restore/dismiss`. Dismiss always clears the banner; Restore
+ * re-reads the plan afterwards, because the server puts back anything it could
+ * not build for a reason that may pass, such as a session limit or an agent that
+ * would not start. The restored sessions arrive as ordinary `session:created`
+ * events, so no extra rendering is needed here.
*
* The banner says that terminal history did not survive, because a restored
* session is a new pane: the conversation continues and the scrollback does not.
diff --git a/src/web/public/styles.css b/src/web/public/styles.css
index b5e4873b..f35bf5cd 100644
--- a/src/web/public/styles.css
+++ b/src/web/public/styles.css
@@ -15277,6 +15277,10 @@ html[data-skin="daylight-blue"] .welcome-btn-tunnel.active:hover {
.reboot-restore-banner-detail {
color: rgba(255, 255, 255, 0.8);
font-weight: 500;
+ /* A flex item will not shrink below its content width at the default
+ `min-width: auto`, so without this the session names push the buttons out of
+ the line between the phone breakpoint and full width. */
+ min-width: 0;
overflow: hidden;
text-overflow: ellipsis;
white-space: nowrap;
diff --git a/src/web/reboot-restore-registry.ts b/src/web/reboot-restore-registry.ts
index 28b4a1d7..0dfab538 100644
--- a/src/web/reboot-restore-registry.ts
+++ b/src/web/reboot-restore-registry.ts
@@ -48,12 +48,16 @@ export class RebootRestoreRegistry {
/** When the boot pass built the plan, in ms since the epoch. */
private builtAt = 0;
/**
- * Bumped by anything that invalidates entries a restore is already holding.
- * A Dismiss arriving mid-restore must win: without this the route's `finally`
- * would put its unspent entries back and resurrect the offer the user just
- * cleared, with a fresh 24-hour life.
+ * Per owner, bumped by anything that invalidates that owner's entries while a
+ * restore is already holding them. A Dismiss arriving mid-restore must win:
+ * without this the route's `finally` would put its unspent entries back and
+ * resurrect the offer the user just cleared, with a fresh 24-hour life.
+ *
+ * Keyed by owner rather than global, because `clear()` is ownership-scoped. A
+ * single counter would let one user's Dismiss discard another user's unspent
+ * entries, and the plan is in-memory, so those offers would be gone for good.
*/
- private generation = 0;
+ private generations = new Map();
/**
* Owners with a restore in flight, between its take and its last pane.
* Keyed by owner so one user's restore does not turn another user's click into
@@ -66,12 +70,17 @@ export class RebootRestoreRegistry {
set(entries: readonly RebootRestoreEntry[]): void {
this.entries = new Map(entries.map((entry) => [entry.sessionId, entry]));
this.builtAt = entries.length > 0 ? Date.now() : 0;
- this.generation += 1;
+ this.bumpAll();
}
- /** The current generation, for a caller that will later return entries. */
- currentGeneration(): number {
- return this.generation;
+ /**
+ * The generations of the owners of `entries`, for a caller that will hand some
+ * of them back later. Pass the result to {@link restore}.
+ */
+ snapshotGenerations(entries: readonly RebootRestoreEntry[]): Map {
+ const snapshot = new Map();
+ for (const entry of entries) snapshot.set(entry.owner, this.generations.get(entry.owner) ?? 0);
+ return snapshot;
}
/**
@@ -117,12 +126,20 @@ export class RebootRestoreRegistry {
* hand is NOT put back, because that one cannot stop being true, and an entry
* the banner keeps re-offering forever is noise only Dismiss can clear.
*/
- restore(entries: readonly RebootRestoreEntry[], generation?: number): void {
- // A dismiss (or a fresh boot plan) since the caller took these entries means
- // they are no longer wanted back.
- if (generation !== undefined && generation !== this.generation) return;
- for (const entry of entries) this.entries.set(entry.sessionId, entry);
- if (entries.length > 0 && this.builtAt === 0) this.builtAt = Date.now();
+ restore(entries: readonly RebootRestoreEntry[], generations?: ReadonlyMap): void {
+ let added = 0;
+ for (const entry of entries) {
+ // A dismiss (or a fresh boot plan) for THIS entry's owner since the caller
+ // took it means it is no longer wanted back. Another owner's dismiss is
+ // none of this entry's business.
+ if (generations) {
+ const taken = generations.get(entry.owner);
+ if (taken !== undefined && taken !== (this.generations.get(entry.owner) ?? 0)) continue;
+ }
+ this.entries.set(entry.sessionId, entry);
+ added += 1;
+ }
+ if (added > 0 && this.builtAt === 0) this.builtAt = Date.now();
}
/** Drop the entries a viewer can see. Returns how many went. */
@@ -130,8 +147,14 @@ export class RebootRestoreRegistry {
const removable = [...this.entries.values()].filter((entry) => canAccess(entry.owner));
for (const entry of removable) this.entries.delete(entry.sessionId);
if (this.entries.size === 0) this.builtAt = 0;
- // Any restore currently in flight must not put its entries back afterwards.
- this.generation += 1;
+ // A restore in flight for these owners must not put their entries back. The
+ // in-flight owners are the ones that matter and the ones the plan can no
+ // longer name: `take()` has already removed their entries, so a dismiss that
+ // lands mid-restore sees nothing of theirs to remove. The bump is limited to
+ // owners this caller could see, so it cannot reach anyone else's restore.
+ const invalidated = new Set(removable.map((entry) => entry.owner));
+ for (const owner of this.spending) if (canAccess(owner)) invalidated.add(owner);
+ for (const owner of invalidated) this.bump(owner);
return removable.length;
}
@@ -155,11 +178,25 @@ export class RebootRestoreRegistry {
this.entries.clear();
this.builtAt = 0;
this.spending.clear();
- this.generation += 1;
+ this.generations.clear();
+ }
+
+ private bump(owner: string | undefined): void {
+ this.generations.set(owner, (this.generations.get(owner) ?? 0) + 1);
+ }
+
+ /** Invalidate every owner's in-flight returns, including owners not yet seen. */
+ private bumpAll(): void {
+ for (const owner of new Set([...this.entries.values()].map((entry) => entry.owner))) this.bump(owner);
+ for (const owner of [...this.generations.keys()]) this.bump(owner);
}
private dropIfExpired(): void {
if (this.builtAt > 0 && Date.now() - this.builtAt > PLAN_TTL_MS) {
+ // Bump before clearing, while the owners are still known: a restore that
+ // took entries just before the expiry must not hand them back afterwards
+ // and give an expired plan another full day of life.
+ this.bumpAll();
this.entries.clear();
this.builtAt = 0;
}
diff --git a/src/web/routes/reboot-restore-routes.ts b/src/web/routes/reboot-restore-routes.ts
index 636f2cd5..da742cc4 100644
--- a/src/web/routes/reboot-restore-routes.ts
+++ b/src/web/routes/reboot-restore-routes.ts
@@ -92,8 +92,8 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
if (!rebootRestoreRegistry.beginSpending(owner)) {
return reply.code(409).send(createErrorResponse(ApiErrorCode.CONFLICT, 'A reboot restore is already running'));
}
- const generation = rebootRestoreRegistry.currentGeneration();
const taken = rebootRestoreRegistry.take(canAccess, body.sessionIds);
+ const generations = rebootRestoreRegistry.snapshotGenerations(taken);
// Entries nothing built a pane for, returned to the plan on every exit path
// including a throw. Without this a failure between here and the loop would
// spend the offer and rebuild nothing, and the plan cannot be rebuilt.
@@ -184,16 +184,20 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
});
await ctx.addSession(session);
- await ctx.setupSessionListeners(session);
- // Shapes the pane, so it has to land before the CLI process starts.
+ // Before the listeners, because setupSessionListeners() reads the
+ // image-watcher flag this phase restores; before the spawn, because the
+ // custom-model environment and the nice priority shape the process.
await ctx.reapplyPersistedSessionState(session, saved, 'before-spawn');
+ await ctx.setupSessionListeners(session);
await session.startInteractive();
// The session's own history, applied only once the pane exists: on a
// failed start these totals would belong to a session that never ran.
- // Both halves precede the first persist, because a constructed session
- // carries none of this and `toState()` is written wholesale, so
- // persisting first would replace the fuller record with the reduced one
- // and drop the pin that keeps it from being pruned.
+ // Both halves precede the route's OWN persist, which matters because a
+ // constructed session carries none of this and `toState()` is written
+ // wholesale, so persisting first would replace the fuller record with
+ // the reduced one and drop the pin that keeps it from being pruned. A
+ // listener-driven persist can still land inside the debounce window
+ // while the pane starts; the write below repairs the record.
await ctx.reapplyPersistedSessionState(session, saved, 'after-spawn');
ctx.persistSessionState(session);
@@ -248,8 +252,9 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
} finally {
// Anything that never became a pane goes back on offer, including after a
// throw, so a transient failure costs a retry rather than the whole plan.
- // Passing the generation makes a Dismiss that landed mid-restore win.
- rebootRestoreRegistry.restore([...unspent], generation);
+ // Passing the generations makes a Dismiss that landed mid-restore win, for
+ // the owners it actually covered.
+ rebootRestoreRegistry.restore([...unspent], generations);
rebootRestoreRegistry.endSpending(owner);
}
});
diff --git a/src/web/server.ts b/src/web/server.ts
index 474d8cd2..095b094d 100644
--- a/src/web/server.ts
+++ b/src/web/server.ts
@@ -2923,8 +2923,9 @@ export class WebServer extends EventEmitter {
* Split in two phases because the two halves have opposite timing needs:
*
* - `before-spawn` shapes the pane itself, so it has to land before the CLI
- * process starts. The custom-model selection is an environment injection and
- * the nice priority is applied to the spawn.
+ * process starts, and before `setupSessionListeners()`, which reads the
+ * image-watcher flag. The custom-model selection is an environment injection
+ * and the nice priority is applied to the spawn.
* - `after-spawn` is the session's own accumulated history. It must NOT land
* on a session whose pane failed to start: the totals would then belong to a
* session that never ran, and any later cleanup would add them to the
@@ -2952,6 +2953,10 @@ export class WebServer extends EventEmitter {
if (saved.niceEnabled !== undefined || saved.niceValue !== undefined) {
session.setNice({ enabled: saved.niceEnabled, niceValue: saved.niceValue });
}
+ // `setupSessionListeners()` READS this flag to decide whether to start the
+ // watcher, so setting it later would leave the session reporting the feature
+ // as on with nothing watching.
+ if (saved.imageWatcherEnabled !== undefined) session.imageWatcherEnabled = saved.imageWatcherEnabled;
return;
}
@@ -2974,7 +2979,6 @@ export class WebServer extends EventEmitter {
});
}
if (saved.color) session.setColor(saved.color);
- if (saved.imageWatcherEnabled !== undefined) session.imageWatcherEnabled = saved.imageWatcherEnabled;
if (saved.flickerFilterEnabled !== undefined) session.flickerFilterEnabled = saved.flickerFilterEnabled;
}
@@ -2989,33 +2993,61 @@ export class WebServer extends EventEmitter {
* WORKING DIRECTORY, which belongs to the workspace rather than to this session
* and may hold another live session's pasted images.
*
- * This undoes only what the failed construction did: the map entry, the tab
- * layout slot `registerSessionWithLayout()` took, and any pane the CLI launch
- * managed to create before it threw. The persisted record is left exactly as it
- * was, so the session stays restorable on the next attempt.
+ * Everything else `_doCleanupSession()` does, this has to do as well. It is the
+ * inverse of `registerSessionWithLayout()` plus `setupSessionListeners()`, and
+ * every registration those two make has to come back out — above all
+ * `sessionListenerRefs`, whose presence makes `setupSessionListeners()` return
+ * early. Leaving that entry behind is worse than the leak this function exists
+ * to prevent: the retry reuses the same session id, wires no listeners at all,
+ * and the user gets a tab that never shows output.
+ *
+ * The persisted record, the lifetime totals, the stored Ralph state and the
+ * workspace's own files are left exactly as they were, so the session stays
+ * restorable on the next attempt.
*/
async discardPartiallyBuiltSession(sessionId: string): Promise {
const session = this.sessions.get(sessionId);
if (!session) return;
this.sessions.delete(sessionId);
+
+ // --- the inverse of setupSessionListeners(), in its order ---
+ const summaryTracker = this.runSummaryTrackers.get(sessionId);
+ if (summaryTracker) {
+ summaryTracker.stop();
+ this.runSummaryTrackers.delete(sessionId);
+ }
+ // An fs.watch on the workspace (or on @fix_plan.md) that nothing else closes.
+ session.ralphTracker.stopWatchingFixPlan();
+ // An FSWatcher on the workspace, likewise.
+ imageWatcher.unwatchSession(sessionId);
+ const listeners = this.sessionListenerRefs.get(sessionId);
+ if (listeners) {
+ detachSessionListeners(session, listeners);
+ this.sessionListenerRefs.delete(sessionId);
+ }
+
+ // --- the inverse of the construction itself ---
this.sse.cleanupSessionBatches(sessionId);
this.persistDeb.cancelKey(sessionId);
+ fileStreamManager.closeSessionStreams(sessionId);
+ // The per-session custom-model config dir carries the endpoint's API key, and
+ // `before-spawn` may already have written it. Nothing else would ever remove
+ // it: the stale sweep only touches state.json. A retry rewrites it.
+ removeConfigDir(customModelConfigDir(sessionId));
try {
session.removeAllListeners();
- await session.stop?.();
+ await session.stop(true);
} catch (err) {
console.warn(`[Server] stopping a partially built session failed: ${getErrorMessage(err)}`);
}
- try {
- await this.mux.killSession(sessionId);
- } catch {
- // The pane may never have been created; nothing to kill is the normal case.
- }
try {
await this.tabLayouts.sessionsRemoved([{ id: sessionId, owner: session.owner }]);
} catch (err) {
console.warn(`[Server] releasing the tab layout slot failed: ${getErrorMessage(err)}`);
}
+ // Any `session:updated` the half-built session emitted before it failed left a
+ // tab on every other open board, and the client's handler is an upsert.
+ this.broadcast(SseEvent.SessionDeleted, { id: sessionId });
}
private async restoreMuxSessions(): Promise {
diff --git a/test/discard-partially-built-session.test.ts b/test/discard-partially-built-session.test.ts
new file mode 100644
index 00000000..7232e969
--- /dev/null
+++ b/test/discard-partially-built-session.test.ts
@@ -0,0 +1,113 @@
+/**
+ * `WebServer.discardPartiallyBuiltSession()` against the real server object.
+ *
+ * The reboot-restore route calls this when a rebuild registers a session and
+ * then fails to start its pane. It has to be the exact inverse of
+ * `registerSessionWithLayout()` plus `setupSessionListeners()`, and it must NOT
+ * be the user-initiated delete: banking the session's token totals, demoting a
+ * pinned record or deleting the workspace's files would all be wrong for a
+ * session that never ran.
+ *
+ * These tests drive the real method rather than the route, because the route
+ * tests run against a mock context whose `discardPartiallyBuiltSession` is a
+ * one-line stub — an earlier version of this function left four registrations
+ * behind and every route test still passed.
+ *
+ * The retry assertion is the important one. `setupSessionListeners()` returns
+ * early when `sessionListenerRefs` still holds the session id, so a discard that
+ * leaves that entry makes the next attempt wire nothing at all, and the user
+ * gets a tab that never shows output.
+ */
+import { mkdirSync, rmSync } from 'node:fs';
+import { homedir } from 'node:os';
+import { join } from 'node:path';
+import { afterEach, beforeEach, describe, expect, it } from 'vitest';
+
+import { WebServer } from '../src/web/server.js';
+import { Session } from '../src/session.js';
+import { TmuxManager } from '../src/tmux-manager.js';
+
+/** Reach the private collections the discard is responsible for emptying. */
+interface ServerInternals {
+ sessions: Map;
+ sessionListenerRefs: Map;
+ runSummaryTrackers: Map;
+ registerSessionWithLayout(session: Session): Promise;
+ setupSessionListeners(session: Session): Promise;
+ discardPartiallyBuiltSession(sessionId: string): Promise;
+}
+
+const WORKSPACE = join(homedir(), '.codeman-test-discard');
+const SESSION_ID = 'a1b2c3d4e5f60718';
+
+let server: WebServer;
+let internals: ServerInternals;
+let mux: TmuxManager;
+
+function buildSession(): Session {
+ return new Session({
+ id: SESSION_ID,
+ workingDir: WORKSPACE,
+ mode: 'claude',
+ name: 'rebuilt session',
+ mux,
+ useMux: true,
+ });
+}
+
+beforeEach(() => {
+ mkdirSync(WORKSPACE, { recursive: true });
+ // Test mode: no port is opened and no CLI is launched.
+ server = new WebServer(0, false, true);
+ internals = server as unknown as ServerInternals;
+ mux = new TmuxManager();
+});
+
+afterEach(async () => {
+ await internals.discardPartiallyBuiltSession(SESSION_ID).catch(() => {});
+ rmSync(WORKSPACE, { recursive: true, force: true });
+});
+
+describe('discarding a session whose pane never started', () => {
+ it('takes the session back out of the server', async () => {
+ const session = buildSession();
+ await internals.registerSessionWithLayout(session);
+ await internals.setupSessionListeners(session);
+ expect(internals.sessions.has(SESSION_ID)).toBe(true);
+
+ await internals.discardPartiallyBuiltSession(SESSION_ID);
+ expect(internals.sessions.has(SESSION_ID)).toBe(false);
+ });
+
+ it('releases the listener registration, so a retry can wire itself again', async () => {
+ const first = buildSession();
+ await internals.registerSessionWithLayout(first);
+ await internals.setupSessionListeners(first);
+ expect(internals.sessionListenerRefs.has(SESSION_ID)).toBe(true);
+
+ await internals.discardPartiallyBuiltSession(SESSION_ID);
+ expect(internals.sessionListenerRefs.has(SESSION_ID)).toBe(false);
+
+ // The retry reuses the id by design. `setupSessionListeners()` returns early
+ // while the refs are still there, so a session built now would run blind:
+ // no terminal output, no status updates, no exit broadcast.
+ const retry = buildSession();
+ await internals.registerSessionWithLayout(retry);
+ await internals.setupSessionListeners(retry);
+ expect(internals.sessionListenerRefs.has(SESSION_ID)).toBe(true);
+ });
+
+ it('stops the run-summary tracker, whose interval would otherwise keep firing', async () => {
+ const session = buildSession();
+ await internals.registerSessionWithLayout(session);
+ await internals.setupSessionListeners(session);
+ expect(internals.runSummaryTrackers.has(SESSION_ID)).toBe(true);
+
+ await internals.discardPartiallyBuiltSession(SESSION_ID);
+ expect(internals.runSummaryTrackers.has(SESSION_ID)).toBe(false);
+ });
+
+ it('does nothing at all for a session it never registered', async () => {
+ await expect(internals.discardPartiallyBuiltSession('never-existed')).resolves.toBeUndefined();
+ });
+});
diff --git a/test/mocks/mock-route-context-completeness.test.ts b/test/mocks/mock-route-context-completeness.test.ts
new file mode 100644
index 00000000..effb98c8
--- /dev/null
+++ b/test/mocks/mock-route-context-completeness.test.ts
@@ -0,0 +1,28 @@
+/**
+ * The mock route context must offer everything the real one does.
+ *
+ * Route tests pass their context as `ctx as never`, and `tsconfig.json` includes
+ * only `src/**`, so no type check ever compares the mock against the ports. A
+ * port that gained a method left this mock missing it twice; both times the
+ * route under test threw a TypeError inside its own catch, and the suite
+ * reported a plausible-looking failure for an unrelated reason.
+ *
+ * So the comparison is made at runtime, against `WebServer.createRouteContext()`
+ * rather than against the port types, which is what keeps it from drifting: the
+ * server's own context object is the thing route modules are really given.
+ */
+import { describe, expect, it } from 'vitest';
+
+import { WebServer } from '../../src/web/server.js';
+import { createMockRouteContext } from './mock-route-context.js';
+
+describe('the mock route context', () => {
+ it('offers every member the real route context does', () => {
+ const server = new WebServer(0, false, true);
+ const real = (server as unknown as { createRouteContext(): Record }).createRouteContext();
+ const mock = createMockRouteContext() as unknown as Record;
+
+ const missing = Object.keys(real).filter((key) => !(key in mock));
+ expect(missing, `mock-route-context.ts is missing: ${missing.join(', ')}`).toEqual([]);
+ });
+});
diff --git a/test/routes/reboot-restore-rebuild-failure.test.ts b/test/routes/reboot-restore-rebuild-failure.test.ts
index f1a35417..9ee93fb4 100644
--- a/test/routes/reboot-restore-rebuild-failure.test.ts
+++ b/test/routes/reboot-restore-rebuild-failure.test.ts
@@ -282,18 +282,62 @@ describe('a failure before any entry is considered', () => {
});
describe('a dismiss that lands while a restore is running', () => {
- it('wins, rather than being undone when the restore hands its entries back', async () => {
- const entries = [offerEntry('a')];
- rebootRestoreRegistry.set(entries);
- const generation = rebootRestoreRegistry.currentGeneration();
- const taken = rebootRestoreRegistry.take(() => true);
- expect(taken).toHaveLength(1);
+ it('wins, rather than being undone when the route hands its entries back', async () => {
+ rebootRestoreRegistry.set([offerEntry('a')]);
+ const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
+ // The user clicks Dismiss while the restore is between its take and its
+ // return. Driven through the ROUTE, so removing the generation argument from
+ // the route would make this fail.
+ (ctx.getWorkspaceHooksEnabled as ReturnType).mockImplementation(async () => {
+ rebootRestoreRegistry.clear(() => true);
+ throw new Error('settings unreadable');
+ });
+ const app = await createHarness(ctx);
- // The user clears the banner while the restore is still working.
- rebootRestoreRegistry.clear(() => true);
- // The restore finishes and tries to put its unspent entry back.
- rebootRestoreRegistry.restore(taken, generation);
+ await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} });
+
+ const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data;
+ expect(left.sessions).toEqual([]);
+ await app.close();
+ });
+
+ it('reaches an in-flight restore the dismisser can see, even once its entries are taken', async () => {
+ const mine = offerEntry('mine', 'alice');
+ rebootRestoreRegistry.set([mine]);
+ expect(rebootRestoreRegistry.beginSpending('alice')).toBe(true);
+ const generations = rebootRestoreRegistry.snapshotGenerations([mine]);
+ const taken = rebootRestoreRegistry.take((owner) => owner === 'alice');
+
+ // The plan is empty now, so a dismiss has nothing of Alice's to remove; the
+ // invalidation has to come from her claimed flight.
+ rebootRestoreRegistry.clear((owner) => owner === 'alice');
+ rebootRestoreRegistry.restore(taken, generations);
+ rebootRestoreRegistry.endSpending('alice');
expect(rebootRestoreRegistry.list(() => true)).toEqual([]);
});
+
+ it('does not reach another owner, whose unspent entries still come back', async () => {
+ const mine = offerEntry('mine', 'alice');
+ const theirs = offerEntry('theirs', 'bob');
+ rebootRestoreRegistry.set([mine, theirs]);
+
+ // Bob is mid-restore, holding his own entry. The claimed flight is what makes
+ // this the interesting case: a dismiss can no longer see Bob's entries in the
+ // plan, so the invalidation has to come from the in-flight set, filtered by
+ // what the dismissing user may access.
+ expect(rebootRestoreRegistry.beginSpending('bob')).toBe(true);
+ const bobsGenerations = rebootRestoreRegistry.snapshotGenerations([theirs]);
+ const bobsTaken = rebootRestoreRegistry.take((owner) => owner === 'bob');
+ expect(bobsTaken.map((e) => e.sessionId)).toEqual(['theirs']);
+
+ // Alice dismisses her own banner meanwhile.
+ rebootRestoreRegistry.clear((owner) => owner === 'alice');
+
+ // Bob's restore finishes and hands his entry back. Alice's dismiss covered
+ // her entries, not his, so his offer survives.
+ rebootRestoreRegistry.restore(bobsTaken, bobsGenerations);
+ rebootRestoreRegistry.endSpending('bob');
+ expect(rebootRestoreRegistry.list(() => true).map((e) => e.sessionId)).toEqual(['theirs']);
+ });
});
From 39976041e0032c1295c08559b1e3627b4c1d16b7 Mon Sep 17 00:00:00 2001
From: Michael Grundberg
Date: Wed, 16 Sep 2026 16:51:00 +0200
Subject: [PATCH 14/28] fix(sessions): let a dismiss reach the entries a
restore is holding
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Fourth review of the reboot-restore branch, and the third to find a defect
in the previous round's fix. This one is the same shape as its predecessor:
a counter keyed on one thing, compared against a set keyed on another.
The generation counter was indexed by the entry's owner, while the in-flight
set holds the caller doing the restoring. Those are the same person exactly
when a user restores their own sessions, which is every case the tests
covered. The route deliberately supports the other case: an admin may spend
another user's entries. So when an admin restored Bob's sessions and Bob
dismissed the banner, nothing matched, the entries came back, and a plan Bob
had explicitly dismissed was re-armed for another twenty-four hours.
Rather than reconcile the two key spaces, the counter is gone. `take()` now
parks the entries it hands out, remembering which caller is spending them,
and they stay parked until that restore ends. A dismiss filters the parked
entries by `canAccess(entry.owner)` — the same predicate it already applies
to the plan — so it reaches them wherever they are. `releaseFlight()` puts
back only what is still parked. Expiry and a fresh boot plan unpark
everything, for the same reason. There is one key space now, the entry's
owner, and the spender is only ever used to tell two concurrent flights
apart. That removes `generations`, `snapshotGenerations()`, `bump()`,
`bumpAll()` and the argument threaded through the route.
The discard grew the teardown it still lacked. A rebuild can fail after
startInteractive() resolved, and a restored workspace still carries
Codeman's hooks, so the CLI can post a hook event within milliseconds; the
transcript watcher that starts from it, the attachment registry, the wait
registry and the approvals inbox all outlive the listeners and would meet
the retry, which reuses the session id by design. Its steps also run in
reverse order now, so no live listener can reach a tracker that has already
stopped, and the mux kill has its own guard, because stop() kills the pane
in its last block after destroying four trackers.
Tests. The run-summary test named an interval and asserted a map entry, so
dropping stop() left it green; it now spies on stop(). Nothing pinned that
before-spawn must precede setupSessionListeners, which reads the flag that
phase restores, so swapping the two lines was silent; the ordering test now
includes the listener setup. The retry assertion was a tautology and now
asserts a different refs object. Both strengthened tests were verified by
reverting their fix. Two new tests cover the admin-restores-another-owner
cases this round was about. The server in the discard test is built once and
stopped, since its constructor registers handlers on module-level watchers,
and the workspace is removed through safeRmHomeTree.
Refs #411
Co-Authored-By: Claude Opus 5 (1M context)
---
src/web/reboot-restore-registry.ts | 99 +++++++++----------
src/web/routes/reboot-restore-routes.ts | 9 +-
src/web/server.ts | 39 ++++++--
test/discard-partially-built-session.test.ts | 36 +++++--
test/reboot-restore.test.ts | 12 +--
.../reboot-restore-rebuild-failure.test.ts | 67 +++++++++----
6 files changed, 159 insertions(+), 103 deletions(-)
diff --git a/src/web/reboot-restore-registry.ts b/src/web/reboot-restore-registry.ts
index 0dfab538..94a7c63f 100644
--- a/src/web/reboot-restore-registry.ts
+++ b/src/web/reboot-restore-registry.ts
@@ -48,16 +48,18 @@ export class RebootRestoreRegistry {
/** When the boot pass built the plan, in ms since the epoch. */
private builtAt = 0;
/**
- * Per owner, bumped by anything that invalidates that owner's entries while a
- * restore is already holding them. A Dismiss arriving mid-restore must win:
- * without this the route's `finally` would put its unspent entries back and
- * resurrect the offer the user just cleared, with a fresh 24-hour life.
+ * Entries handed to a restore that has not finished, by session id, each
+ * remembering which caller is spending it.
*
- * Keyed by owner rather than global, because `clear()` is ownership-scoped. A
- * single counter would let one user's Dismiss discard another user's unspent
- * entries, and the plan is in-memory, so those offers would be gone for good.
+ * A taken entry is still part of the offer until its restore resolves it, so
+ * it has to stay reachable by everything that can invalidate an offer. Holding
+ * the entries themselves — rather than a counter to compare against later —
+ * means `clear()` filters them by the SAME `canAccess(entry.owner)` predicate
+ * it already applies to the plan. A counter cannot do that, because the caller
+ * spending an entry need not be its owner: an admin may restore another user's
+ * sessions, and then the spender and the owner are different keys.
*/
- private generations = new Map();
+ private parked = new Map();
/**
* Owners with a restore in flight, between its take and its last pane.
* Keyed by owner so one user's restore does not turn another user's click into
@@ -70,17 +72,8 @@ export class RebootRestoreRegistry {
set(entries: readonly RebootRestoreEntry[]): void {
this.entries = new Map(entries.map((entry) => [entry.sessionId, entry]));
this.builtAt = entries.length > 0 ? Date.now() : 0;
- this.bumpAll();
- }
-
- /**
- * The generations of the owners of `entries`, for a caller that will hand some
- * of them back later. Pass the result to {@link restore}.
- */
- snapshotGenerations(entries: readonly RebootRestoreEntry[]): Map {
- const snapshot = new Map();
- for (const entry of entries) snapshot.set(entry.owner, this.generations.get(entry.owner) ?? 0);
- return snapshot;
+ // A fresh boot plan supersedes anything an in-flight restore still holds.
+ this.parked.clear();
}
/**
@@ -104,7 +97,11 @@ export class RebootRestoreRegistry {
*
* @param sessionIds The ids to spend, or undefined for every visible entry.
*/
- take(canAccess: (owner: string | undefined) => boolean, sessionIds?: readonly string[]): RebootRestoreEntry[] {
+ take(
+ canAccess: (owner: string | undefined) => boolean,
+ sessionIds: readonly string[] | undefined,
+ spender: string | undefined
+ ): RebootRestoreEntry[] {
this.dropIfExpired();
const wanted = sessionIds ? new Set(sessionIds) : undefined;
const taken: RebootRestoreEntry[] = [];
@@ -112,6 +109,9 @@ export class RebootRestoreRegistry {
if (wanted && !wanted.has(entry.sessionId)) continue;
if (!canAccess(entry.owner)) continue;
this.entries.delete(entry.sessionId);
+ // Parked rather than forgotten: until this restore resolves the entry, a
+ // dismiss still has to be able to reach and cancel it.
+ this.parked.set(entry.sessionId, { entry, spender });
taken.push(entry);
}
return taken;
@@ -126,18 +126,19 @@ export class RebootRestoreRegistry {
* hand is NOT put back, because that one cannot stop being true, and an entry
* the banner keeps re-offering forever is noise only Dismiss can clear.
*/
- restore(entries: readonly RebootRestoreEntry[], generations?: ReadonlyMap): void {
+ releaseFlight(spender: string | undefined, keep: readonly RebootRestoreEntry[]): void {
+ const wanted = new Set(keep.map((entry) => entry.sessionId));
let added = 0;
- for (const entry of entries) {
- // A dismiss (or a fresh boot plan) for THIS entry's owner since the caller
- // took it means it is no longer wanted back. Another owner's dismiss is
- // none of this entry's business.
- if (generations) {
- const taken = generations.get(entry.owner);
- if (taken !== undefined && taken !== (this.generations.get(entry.owner) ?? 0)) continue;
+ for (const [sessionId, held] of [...this.parked]) {
+ if (held.spender !== spender) continue;
+ this.parked.delete(sessionId);
+ // Still parked means nothing cancelled it while the restore ran. A dismiss,
+ // an expiry or a fresh boot plan removes it from `parked`, and then it does
+ // not come back however the restore ended.
+ if (wanted.has(sessionId)) {
+ this.entries.set(sessionId, held.entry);
+ added += 1;
}
- this.entries.set(entry.sessionId, entry);
- added += 1;
}
if (added > 0 && this.builtAt === 0) this.builtAt = Date.now();
}
@@ -146,16 +147,17 @@ export class RebootRestoreRegistry {
clear(canAccess: (owner: string | undefined) => boolean): number {
const removable = [...this.entries.values()].filter((entry) => canAccess(entry.owner));
for (const entry of removable) this.entries.delete(entry.sessionId);
+ // Entries a restore is holding are dismissed by the same rule, so a dismiss
+ // that lands mid-restore wins. Judged on the ENTRY's owner, exactly as above,
+ // rather than on who happens to be restoring it.
+ let parkedRemoved = 0;
+ for (const [sessionId, held] of [...this.parked]) {
+ if (!canAccess(held.entry.owner)) continue;
+ this.parked.delete(sessionId);
+ parkedRemoved += 1;
+ }
if (this.entries.size === 0) this.builtAt = 0;
- // A restore in flight for these owners must not put their entries back. The
- // in-flight owners are the ones that matter and the ones the plan can no
- // longer name: `take()` has already removed their entries, so a dismiss that
- // lands mid-restore sees nothing of theirs to remove. The bump is limited to
- // owners this caller could see, so it cannot reach anyone else's restore.
- const invalidated = new Set(removable.map((entry) => entry.owner));
- for (const owner of this.spending) if (canAccess(owner)) invalidated.add(owner);
- for (const owner of invalidated) this.bump(owner);
- return removable.length;
+ return removable.length + parkedRemoved;
}
/**
@@ -176,27 +178,16 @@ export class RebootRestoreRegistry {
/** Test hook: forget everything, including the single-flight claim. */
reset(): void {
this.entries.clear();
+ this.parked.clear();
this.builtAt = 0;
this.spending.clear();
- this.generations.clear();
- }
-
- private bump(owner: string | undefined): void {
- this.generations.set(owner, (this.generations.get(owner) ?? 0) + 1);
- }
-
- /** Invalidate every owner's in-flight returns, including owners not yet seen. */
- private bumpAll(): void {
- for (const owner of new Set([...this.entries.values()].map((entry) => entry.owner))) this.bump(owner);
- for (const owner of [...this.generations.keys()]) this.bump(owner);
}
private dropIfExpired(): void {
if (this.builtAt > 0 && Date.now() - this.builtAt > PLAN_TTL_MS) {
- // Bump before clearing, while the owners are still known: a restore that
- // took entries just before the expiry must not hand them back afterwards
- // and give an expired plan another full day of life.
- this.bumpAll();
+ // A restore that took entries just before the expiry must not hand them
+ // back afterwards and give an expired plan another full day of life.
+ this.parked.clear();
this.entries.clear();
this.builtAt = 0;
}
diff --git a/src/web/routes/reboot-restore-routes.ts b/src/web/routes/reboot-restore-routes.ts
index da742cc4..3516ff19 100644
--- a/src/web/routes/reboot-restore-routes.ts
+++ b/src/web/routes/reboot-restore-routes.ts
@@ -92,8 +92,7 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
if (!rebootRestoreRegistry.beginSpending(owner)) {
return reply.code(409).send(createErrorResponse(ApiErrorCode.CONFLICT, 'A reboot restore is already running'));
}
- const taken = rebootRestoreRegistry.take(canAccess, body.sessionIds);
- const generations = rebootRestoreRegistry.snapshotGenerations(taken);
+ const taken = rebootRestoreRegistry.take(canAccess, body.sessionIds, owner);
// Entries nothing built a pane for, returned to the plan on every exit path
// including a throw. Without this a failure between here and the loop would
// spend the offer and rebuild nothing, and the plan cannot be rebuilt.
@@ -252,9 +251,9 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
} finally {
// Anything that never became a pane goes back on offer, including after a
// throw, so a transient failure costs a retry rather than the whole plan.
- // Passing the generations makes a Dismiss that landed mid-restore win, for
- // the owners it actually covered.
- rebootRestoreRegistry.restore([...unspent], generations);
+ // Ends the flight: entries still parked for it come back if they are in
+ // `unspent`, and a Dismiss that unparked them meanwhile wins.
+ rebootRestoreRegistry.releaseFlight(owner, [...unspent]);
rebootRestoreRegistry.endSpending(owner);
}
});
diff --git a/src/web/server.ts b/src/web/server.ts
index 095b094d..2358b2ff 100644
--- a/src/web/server.ts
+++ b/src/web/server.ts
@@ -3010,26 +3010,42 @@ export class WebServer extends EventEmitter {
if (!session) return;
this.sessions.delete(sessionId);
- // --- the inverse of setupSessionListeners(), in its order ---
- const summaryTracker = this.runSummaryTrackers.get(sessionId);
- if (summaryTracker) {
- summaryTracker.stop();
- this.runSummaryTrackers.delete(sessionId);
- }
- // An fs.watch on the workspace (or on @fix_plan.md) that nothing else closes.
- session.ralphTracker.stopWatchingFixPlan();
- // An FSWatcher on the workspace, likewise.
- imageWatcher.unwatchSession(sessionId);
+ // --- the inverse of setupSessionListeners(), in reverse order ---
+ // Listeners first: while they are attached, one of them can still reach a
+ // tracker this is about to stop.
const listeners = this.sessionListenerRefs.get(sessionId);
if (listeners) {
detachSessionListeners(session, listeners);
this.sessionListenerRefs.delete(sessionId);
}
+ // An FSWatcher on the workspace that nothing else closes.
+ imageWatcher.unwatchSession(sessionId);
+ // An fs.watch on the workspace (or on @fix_plan.md), likewise.
+ session.ralphTracker.stopWatchingFixPlan();
+ const summaryTracker = this.runSummaryTrackers.get(sessionId);
+ if (summaryTracker) {
+ summaryTracker.stop();
+ this.runSummaryTrackers.delete(sessionId);
+ }
+
+ // --- what anything else may have attached to this id in the meantime ---
+ // A rebuild can fail AFTER startInteractive() resolved, and a restored
+ // workspace still carries Codeman's hooks, so the CLI can post a hook event
+ // within milliseconds. Each of these outlives the listeners and would
+ // otherwise meet the retry, which reuses the same session id by design.
+ this.stopTranscriptWatcher(sessionId);
+ attachmentRegistry.clearSession(sessionId);
+ sessionWaits.notifySignal(sessionId, 'exit');
+ sessionWaits.cancelAll(sessionId);
+ approvalInbox.resolveForSession(sessionId, 'session_ended');
// --- the inverse of the construction itself ---
this.sse.cleanupSessionBatches(sessionId);
this.persistDeb.cancelKey(sessionId);
fileStreamManager.closeSessionStreams(sessionId);
+ // `lastRecordedTokens` is deliberately NOT deleted: the `after-spawn` phase
+ // seeds it as the daily-usage baseline for these restored totals, and the
+ // retry reuses the id, so dropping it would count them as new usage.
// The per-session custom-model config dir carries the endpoint's API key, and
// `before-spawn` may already have written it. Nothing else would ever remove
// it: the stale sweep only touches state.json. A retry rewrites it.
@@ -3039,6 +3055,9 @@ export class WebServer extends EventEmitter {
await session.stop(true);
} catch (err) {
console.warn(`[Server] stopping a partially built session failed: ${getErrorMessage(err)}`);
+ // `stop()` kills the mux session in its last block, after destroying its
+ // trackers, so a throw on the way there leaves the pane running.
+ await this.mux.killSession(sessionId).catch(() => {});
}
try {
await this.tabLayouts.sessionsRemoved([{ id: sessionId, owner: session.owner }]);
diff --git a/test/discard-partially-built-session.test.ts b/test/discard-partially-built-session.test.ts
index 7232e969..012009f4 100644
--- a/test/discard-partially-built-session.test.ts
+++ b/test/discard-partially-built-session.test.ts
@@ -18,10 +18,11 @@
* leaves that entry makes the next attempt wire nothing at all, and the user
* gets a tab that never shows output.
*/
-import { mkdirSync, rmSync } from 'node:fs';
+import { mkdirSync } from 'node:fs';
import { homedir } from 'node:os';
import { join } from 'node:path';
-import { afterEach, beforeEach, describe, expect, it } from 'vitest';
+import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest';
+import { safeRmHomeTree } from './mocks/test-helpers.js';
import { WebServer } from '../src/web/server.js';
import { Session } from '../src/session.js';
@@ -55,9 +56,11 @@ function buildSession(): Session {
});
}
-beforeEach(() => {
+beforeAll(() => {
mkdirSync(WORKSPACE, { recursive: true });
- // Test mode: no port is opened and no CLI is launched.
+ // Test mode: no port is opened and no CLI is launched. One server for the file,
+ // stopped at the end: the constructor registers handlers on the module-level
+ // image, subagent, team and workflow watchers, and only stop() removes them.
server = new WebServer(0, false, true);
internals = server as unknown as ServerInternals;
mux = new TmuxManager();
@@ -65,7 +68,11 @@ beforeEach(() => {
afterEach(async () => {
await internals.discardPartiallyBuiltSession(SESSION_ID).catch(() => {});
- rmSync(WORKSPACE, { recursive: true, force: true });
+});
+
+afterAll(async () => {
+ await server.stop().catch(() => {});
+ safeRmHomeTree(WORKSPACE);
});
describe('discarding a session whose pane never started', () => {
@@ -85,25 +92,36 @@ describe('discarding a session whose pane never started', () => {
await internals.setupSessionListeners(first);
expect(internals.sessionListenerRefs.has(SESSION_ID)).toBe(true);
+ const firstRefs = internals.sessionListenerRefs.get(SESSION_ID);
await internals.discardPartiallyBuiltSession(SESSION_ID);
expect(internals.sessionListenerRefs.has(SESSION_ID)).toBe(false);
// The retry reuses the id by design. `setupSessionListeners()` returns early
- // while the refs are still there, so a session built now would run blind:
- // no terminal output, no status updates, no exit broadcast.
+ // while the refs are still there, so a session built now would run blind: no
+ // terminal output, no status updates, no exit broadcast. Asserting a DIFFERENT
+ // refs object is what distinguishes wiring the retry from finding the corpse
+ // of the first attempt still in place.
const retry = buildSession();
await internals.registerSessionWithLayout(retry);
await internals.setupSessionListeners(retry);
- expect(internals.sessionListenerRefs.has(SESSION_ID)).toBe(true);
+ const retryRefs = internals.sessionListenerRefs.get(SESSION_ID);
+ expect(retryRefs).toBeDefined();
+ expect(retryRefs).not.toBe(firstRefs);
});
it('stops the run-summary tracker, whose interval would otherwise keep firing', async () => {
const session = buildSession();
await internals.registerSessionWithLayout(session);
await internals.setupSessionListeners(session);
- expect(internals.runSummaryTrackers.has(SESSION_ID)).toBe(true);
+ const tracker = internals.runSummaryTrackers.get(SESSION_ID) as { stop: () => void };
+ expect(tracker).toBeDefined();
+ // Dropping the map entry is not enough: the tracker arms a setInterval in its
+ // constructor, and only stop() clears it, so a discard that merely forgot the
+ // entry would leave the timer running for the life of the process.
+ const stopped = vi.spyOn(tracker, 'stop');
await internals.discardPartiallyBuiltSession(SESSION_ID);
+ expect(stopped).toHaveBeenCalled();
expect(internals.runSummaryTrackers.has(SESSION_ID)).toBe(false);
});
diff --git a/test/reboot-restore.test.ts b/test/reboot-restore.test.ts
index 76506358..aaf9a83e 100644
--- a/test/reboot-restore.test.ts
+++ b/test/reboot-restore.test.ts
@@ -198,15 +198,15 @@ describe('the plan the banner spends', () => {
it('hands an entry to the first caller and nothing to the second', () => {
const registry = new RebootRestoreRegistry();
registry.set([entryFor('a'), entryFor('b')]);
- expect(registry.take(all).map((e) => e.sessionId)).toEqual(['a', 'b']);
+ expect(registry.take(all, undefined, undefined).map((e) => e.sessionId)).toEqual(['a', 'b']);
// The double-click: two panes on one conversation is what this prevents.
- expect(registry.take(all)).toEqual([]);
+ expect(registry.take(all, undefined, undefined)).toEqual([]);
});
it('spends only the ids a caller asked for', () => {
const registry = new RebootRestoreRegistry();
registry.set([entryFor('a'), entryFor('b')]);
- expect(registry.take(all, ['b']).map((e) => e.sessionId)).toEqual(['b']);
+ expect(registry.take(all, ['b'], undefined).map((e) => e.sessionId)).toEqual(['b']);
expect(registry.list(all).map((e) => e.sessionId)).toEqual(['a']);
});
@@ -215,7 +215,7 @@ describe('the plan the banner spends', () => {
registry.set([entryFor('mine', 'alice'), entryFor('theirs', 'bob')]);
const asAlice = (owner: string | undefined) => owner === 'alice';
expect(registry.list(asAlice).map((e) => e.sessionId)).toEqual(['mine']);
- expect(registry.take(asAlice).map((e) => e.sessionId)).toEqual(['mine']);
+ expect(registry.take(asAlice, undefined, 'alice').map((e) => e.sessionId)).toEqual(['mine']);
// Bob's entry is still on offer for Bob.
expect(registry.list(() => true).map((e) => e.sessionId)).toEqual(['theirs']);
});
@@ -223,8 +223,8 @@ describe('the plan the banner spends', () => {
it('puts back an entry that no pane was created for', () => {
const registry = new RebootRestoreRegistry();
registry.set([entryFor('a')]);
- const taken = registry.take(all);
- registry.restore(taken);
+ const taken = registry.take(all, undefined, undefined);
+ registry.releaseFlight(undefined, taken);
expect(registry.list(all).map((e) => e.sessionId)).toEqual(['a']);
});
diff --git a/test/routes/reboot-restore-rebuild-failure.test.ts b/test/routes/reboot-restore-rebuild-failure.test.ts
index 9ee93fb4..7a69e30d 100644
--- a/test/routes/reboot-restore-rebuild-failure.test.ts
+++ b/test/routes/reboot-restore-rebuild-failure.test.ts
@@ -183,12 +183,23 @@ describe('a rebuild that succeeds', () => {
callOrder.push(`reapply:${phase}`);
}
);
+ (ctx.setupSessionListeners as ReturnType).mockImplementation(async () => {
+ callOrder.push('setupSessionListeners');
+ });
const app = await createHarness(ctx);
await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} });
- // The custom-model environment has to reach the process; the token totals
- // must not land on a session whose pane never started.
- expect(callOrder).toEqual(['reapply:before-spawn', 'startInteractive', 'reapply:after-spawn']);
+ // `setupSessionListeners()` READS the image-watcher flag that `before-spawn`
+ // restores, so the phase has to precede it or the session comes back
+ // reporting the watcher as on with nothing watching. The custom-model
+ // environment has to reach the process, and the token totals must not land
+ // on a session whose pane never started.
+ expect(callOrder).toEqual([
+ 'reapply:before-spawn',
+ 'setupSessionListeners',
+ 'startInteractive',
+ 'reapply:after-spawn',
+ ]);
await app.close();
});
@@ -282,6 +293,32 @@ describe('a failure before any entry is considered', () => {
});
describe('a dismiss that lands while a restore is running', () => {
+ it('wins when an admin is restoring the entries and their owner dismisses', async () => {
+ const theirs = offerEntry('theirs', 'bob');
+ rebootRestoreRegistry.set([theirs]);
+ // An admin may spend another user's entries, so the caller doing the restore
+ // and the owner of what is being restored are different people.
+ const taken = rebootRestoreRegistry.take(() => true, undefined, 'admin');
+ expect(taken.map((e) => e.sessionId)).toEqual(['theirs']);
+
+ // Bob dismisses his own banner. Nothing of his is in the plan any more, and
+ // the restore is running under a different name than his.
+ rebootRestoreRegistry.clear((owner) => owner === 'bob');
+ rebootRestoreRegistry.releaseFlight('admin', taken);
+
+ expect(rebootRestoreRegistry.list(() => true)).toEqual([]);
+ });
+
+ it('wins when an admin dismisses everything mid-restore', async () => {
+ rebootRestoreRegistry.set([offerEntry('theirs', 'bob')]);
+ const taken = rebootRestoreRegistry.take(() => true, undefined, 'admin');
+
+ rebootRestoreRegistry.clear(() => true);
+ rebootRestoreRegistry.releaseFlight('admin', taken);
+
+ expect(rebootRestoreRegistry.list(() => true)).toEqual([]);
+ });
+
it('wins, rather than being undone when the route hands its entries back', async () => {
rebootRestoreRegistry.set([offerEntry('a')]);
const ctx = createMockRouteContext({ workspaceHooksEnabled: false });
@@ -304,15 +341,13 @@ describe('a dismiss that lands while a restore is running', () => {
it('reaches an in-flight restore the dismisser can see, even once its entries are taken', async () => {
const mine = offerEntry('mine', 'alice');
rebootRestoreRegistry.set([mine]);
- expect(rebootRestoreRegistry.beginSpending('alice')).toBe(true);
- const generations = rebootRestoreRegistry.snapshotGenerations([mine]);
- const taken = rebootRestoreRegistry.take((owner) => owner === 'alice');
+ const taken = rebootRestoreRegistry.take((owner) => owner === 'alice', undefined, 'alice');
+ expect(taken).toHaveLength(1);
- // The plan is empty now, so a dismiss has nothing of Alice's to remove; the
- // invalidation has to come from her claimed flight.
+ // The plan is empty now, so the dismiss has nothing of Alice's left in the
+ // plan; it has to reach the entry the restore is holding.
rebootRestoreRegistry.clear((owner) => owner === 'alice');
- rebootRestoreRegistry.restore(taken, generations);
- rebootRestoreRegistry.endSpending('alice');
+ rebootRestoreRegistry.releaseFlight('alice', taken);
expect(rebootRestoreRegistry.list(() => true)).toEqual([]);
});
@@ -322,13 +357,8 @@ describe('a dismiss that lands while a restore is running', () => {
const theirs = offerEntry('theirs', 'bob');
rebootRestoreRegistry.set([mine, theirs]);
- // Bob is mid-restore, holding his own entry. The claimed flight is what makes
- // this the interesting case: a dismiss can no longer see Bob's entries in the
- // plan, so the invalidation has to come from the in-flight set, filtered by
- // what the dismissing user may access.
- expect(rebootRestoreRegistry.beginSpending('bob')).toBe(true);
- const bobsGenerations = rebootRestoreRegistry.snapshotGenerations([theirs]);
- const bobsTaken = rebootRestoreRegistry.take((owner) => owner === 'bob');
+ // Bob is mid-restore, holding his own entry.
+ const bobsTaken = rebootRestoreRegistry.take((owner) => owner === 'bob', undefined, 'bob');
expect(bobsTaken.map((e) => e.sessionId)).toEqual(['theirs']);
// Alice dismisses her own banner meanwhile.
@@ -336,8 +366,7 @@ describe('a dismiss that lands while a restore is running', () => {
// Bob's restore finishes and hands his entry back. Alice's dismiss covered
// her entries, not his, so his offer survives.
- rebootRestoreRegistry.restore(bobsTaken, bobsGenerations);
- rebootRestoreRegistry.endSpending('bob');
+ rebootRestoreRegistry.releaseFlight('bob', bobsTaken);
expect(rebootRestoreRegistry.list(() => true).map((e) => e.sessionId)).toEqual(['theirs']);
});
});
From 18ab2ab5950c941bb4f9269c71f3ab7f76811837 Mon Sep 17 00:00:00 2001
From: Michael Grundberg
Date: Thu, 17 Sep 2026 08:26:29 +0200
Subject: [PATCH 15/28] docs(sessions): correct what a failed rebuild is
actually likely to be
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Ran the feature against a real server for the first time, on an isolated
instance, and two claims in the code turned out to be wrong.
A rebuild that fails after the session is registered was documented as
commonly caused by a CLI binary missing from a freshly booted machine's
PATH. It is not: the resolver finds its binary by absolute path, so PATH
never enters into it, and a server started without claude on PATH restored
every session normally. Nor does an un-enterable workspace fail — tmux falls
back to another directory and the pane comes up there. Neither obvious cause
throws, so the discard path is defended rather than expected, and the
comments now say that instead of naming a cause that cannot happen.
The four review rounds that shaped this path all reasoned about a trigger
none of them could test. The path itself is still worth having, since a mux
failure would reach it, but its comments should not claim a likelihood the
machine disagrees with.
Refs #411
Co-Authored-By: Claude Opus 5 (1M context)
---
src/web/routes/reboot-restore-routes.ts | 10 ++++++++--
test/routes/reboot-restore-rebuild-failure.test.ts | 13 +++++++++----
2 files changed, 17 insertions(+), 6 deletions(-)
diff --git a/src/web/routes/reboot-restore-routes.ts b/src/web/routes/reboot-restore-routes.ts
index 3516ff19..3ff27b4f 100644
--- a/src/web/routes/reboot-restore-routes.ts
+++ b/src/web/routes/reboot-restore-routes.ts
@@ -220,8 +220,14 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
// One entry that will not start must not stop the rest of the pass, and
// must not leave a registered session with no pane behind it: by this
// point the session is in `ctx.sessions`, holds a tab-layout slot and has
- // listeners, and the commonest cause is a CLI binary that is not on the
- // PATH of a freshly booted machine.
+ // listeners.
+ //
+ // Reaching this is rarer than it looks, measured against a real server:
+ // the CLI resolver finds its binary by absolute path rather than through
+ // PATH, and tmux falls back to another directory rather than failing when
+ // it cannot enter the workspace, so neither of the two obvious "freshly
+ // booted machine" failures throws. What is left is the mux layer itself
+ // failing, which is why this path is defended rather than expected.
console.error(`[reboot-restore] failed to rebuild ${entry.sessionId}:`, err);
// Not cleanupSession(): that is the user-initiated delete, and it would
// count this session's historical tokens into the lifetime totals, demote
diff --git a/test/routes/reboot-restore-rebuild-failure.test.ts b/test/routes/reboot-restore-rebuild-failure.test.ts
index 7a69e30d..98cf7a17 100644
--- a/test/routes/reboot-restore-rebuild-failure.test.ts
+++ b/test/routes/reboot-restore-rebuild-failure.test.ts
@@ -4,10 +4,15 @@
* The other route test file deliberately uses workspaces that do not exist, so it
* never reaches `new Session()`. This one mocks the `Session` module so the route
* runs its whole construction path — `addSession`, `setupSessionListeners`,
- * `reapplyPersistedSessionState`, `startInteractive` — and then throws where a
- * real one would when the CLI binary is missing from a freshly booted machine's
- * PATH. Without the mock there is no way to exercise that path, which is how the
- * original version of this route shipped a session leak the tests could not see.
+ * `reapplyPersistedSessionState`, `startInteractive` — and then throws.
+ *
+ * The mock is the only way in. Driven against a real server, `startInteractive()`
+ * does not throw for either obvious cause: the CLI resolver finds its binary by
+ * absolute path rather than through PATH, and tmux falls back to another
+ * directory rather than failing when it cannot enter the workspace. A mux-layer
+ * failure is what is left, and it cannot be provoked from a test. Without the
+ * mock this path would go unexercised, which is how the original version of this
+ * route shipped a session leak the tests could not see.
*
* It also covers the session caps, because those too are only reachable once the
* route is actually willing to build something.
From 5f55f9cb65b39716461dd26854dacb91089c2ade Mon Sep 17 00:00:00 2001
From: Michael Grundberg
Date: Thu, 17 Sep 2026 14:11:58 +0200
Subject: [PATCH 16/28] fix(sessions): never let the reboot-restore plan fail
recovery
The plan build runs inside the try that decides whether restoreMuxSessions()
succeeded, so a throw would be caught there, report restoration as failed,
and block the stale cleanup and layout reconciliation that follow. An
optional convenience would then break the recovery it exists to help. It is
guarded on its own now: the correct way for this to fail is an offer nobody
gets.
Refs #411
Co-Authored-By: Claude Opus 5 (1M context)
---
src/web/server.ts | 13 ++++++++++++-
1 file changed, 12 insertions(+), 1 deletion(-)
diff --git a/src/web/server.ts b/src/web/server.ts
index 2358b2ff..e94b8cdf 100644
--- a/src/web/server.ts
+++ b/src/web/server.ts
@@ -3081,7 +3081,18 @@ export class WebServer extends EventEmitter {
// Build the reboot-restore offer HERE: `dead` is only known after
// reconciliation, and the records it reads are pruned by
// `cleanupStaleSessions()` as soon as `finalizeRestoredState()` runs.
- this.planRebootRestoreOffer(dead, alive.length);
+ //
+ // Guarded on its own, because this runs inside the try that decides whether
+ // RECOVERY succeeded. A throw here would otherwise be caught below, report
+ // restoration as failed, and block the stale cleanup and layout
+ // reconciliation that follow — turning an optional convenience into a
+ // failure of the thing it is supposed to help. An offer nobody gets is the
+ // correct way for this to fail.
+ try {
+ this.planRebootRestoreOffer(dead, alive.length);
+ } catch (err) {
+ console.error('[Server] Building the reboot-restore offer failed; continuing recovery:', err);
+ }
if (alive.length > 0 || discovered.length > 0) {
console.log(`[Server] Found ${alive.length + discovered.length} alive mux session(s) from previous run`);
From 5108a24bf0f63eff8d2dd92e661f668a148219c7 Mon Sep 17 00:00:00 2001
From: Michael Grundberg
Date: Thu, 17 Sep 2026 18:46:15 +0200
Subject: [PATCH 17/28] fix(sessions): never restore a session whose agent was
already exited
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Found by a real reboot, which is the first thing to catch it. Typing `/exit`
ends the CLI process and leaves the session record behind, and the
process-exit handler persists `pid: null` with `status: 'idle'` before
anything else runs. By status alone that is indistinguishable from a session
sitting idle when the power went, so the boot pass offered those sessions
back and a click spawned the agents the user had deliberately closed — the
exact case the eligibility rule exists to exclude.
The absent pid is what tells the two apart, and the plan step now refuses a
record without one, under its own `not-running` reason so the boot log says
why. On a healthy board every running session carries a pid; a record with
none describes an agent that is already gone.
Deliberately the conservative direction. A session that somehow persisted no
pid while genuinely running is not offered, and its conversation stays
reachable from the Resume list, which is where every session would be
without this feature. The opposite error spawns processes nobody asked for.
Ark0N/Codeman#446 covers the dead panes those exits leave behind, but this
does not wait on it: the rule belongs here whether or not the record's shape
changes later.
Refs #411
Co-Authored-By: Claude Opus 5 (1M context)
---
src/reboot-restore.ts | 25 ++++++++++++++++++++++++-
test/reboot-restore.test.ts | 27 +++++++++++++++++++++++++++
2 files changed, 51 insertions(+), 1 deletion(-)
diff --git a/src/reboot-restore.ts b/src/reboot-restore.ts
index e226138c..d295a34d 100644
--- a/src/reboot-restore.ts
+++ b/src/reboot-restore.ts
@@ -19,6 +19,13 @@
* touching its status, so a pinned session a reboot killed still reads `idle` or
* `busy` and stays eligible.
*
+ * Ending the AGENT rather than the session leaves a third shape, and it is the one
+ * a real reboot caught this module getting wrong. `/exit` ends the CLI process
+ * while the session record survives, and the process-exit handler persists
+ * `pid: null` with `status: 'idle'` — indistinguishable by status from a session
+ * that was merely idle when the power went. The absent pid is what tells them
+ * apart, so a record without one is refused.
+ *
* @dependencies types (SessionState), config/cli-registry
* @consumedby web/server (plan build at boot), web/routes/reboot-restore-routes
*
@@ -89,7 +96,7 @@ export function resolveResumeConversationId(state: SessionState): string {
/**
* Why one session was passed over. Reported for logging and shown to the user.
*
- * The first six are decided before anything is built. `capacity-reached` and
+ * The first seven are decided before anything is built. `capacity-reached` and
* `rebuild-failed` can only happen once a click is spending the plan, and they
* are the two the banner must not confuse with a missing workspace: one means
* "try again after closing something", the other means the CLI would not start.
@@ -99,6 +106,7 @@ export interface RebootRestoreRejection {
reason:
| 'no-persisted-record'
| 'intentionally-ended'
+ | 'not-running'
| 'respawn-blocked'
| 'remote-or-docker'
| 'unsupported-mode'
@@ -163,6 +171,21 @@ export function planRebootRestore(
skipped.push({ sessionId, reason: 'intentionally-ended' });
continue;
}
+ if (state.pid === null || state.pid === undefined) {
+ // The agent had already exited when the machine went down: `/exit` ends the
+ // process, and its exit handler persists `pid: null` with `status: 'idle'`
+ // before anything else can. Status alone cannot tell that apart from a
+ // session that was simply sitting idle when the power went, so without this
+ // a reboot restore spawns the agents the user deliberately closed — the
+ // exact case the eligibility rule exists to exclude.
+ //
+ // A heuristic, and deliberately the conservative one. A session that somehow
+ // persisted no pid while genuinely running is not offered, and its
+ // conversation stays reachable from the Resume list, which is where every
+ // session would be without this feature.
+ skipped.push({ sessionId, reason: 'not-running' });
+ continue;
+ }
if (state.respawnBlocked === true) {
// The crash-loop breaker tripped on this pane. Re-creating it restarts the loop.
skipped.push({ sessionId, reason: 'respawn-blocked' });
diff --git a/test/reboot-restore.test.ts b/test/reboot-restore.test.ts
index aaf9a83e..5ed989fd 100644
--- a/test/reboot-restore.test.ts
+++ b/test/reboot-restore.test.ts
@@ -39,6 +39,7 @@ const NOW = 1_760_000_000_000;
function persistedSession(overrides: Partial & { id: string }): SessionState {
return {
+ // A live agent's record carries its process id; `/exit` persists null instead.
pid: 99999,
status: 'idle',
workingDir: '/tmp/spike',
@@ -137,6 +138,32 @@ describe('which dead sessions may be rebuilt', () => {
});
});
+describe('a session whose agent had already exited', () => {
+ it('is refused, because `/exit` leaves the record reading idle with no pid', () => {
+ // What the process-exit handler persists: the CLI is gone, the record is not,
+ // and its status is indistinguishable from a session that was merely idle.
+ const persisted = { exited: persistedSession({ id: 'exited', status: 'idle', pid: null }) };
+ const plan = planRebootRestore(['exited'], persisted, () => true);
+ expect(plan.restore).toEqual([]);
+ expect(plan.skipped).toEqual([{ sessionId: 'exited', reason: 'not-running' }]);
+ });
+
+ it('still restores the session beside it that was running when the power went', () => {
+ const persisted = {
+ exited: persistedSession({ id: 'exited', pid: null }),
+ running: persistedSession({ id: 'running', pid: 4242 }),
+ };
+ const plan = planRebootRestore(['exited', 'running'], persisted, () => true);
+ expect(plan.restore.map((entry) => entry.sessionId)).toEqual(['running']);
+ expect(plan.skipped.map((s) => s.reason)).toEqual(['not-running']);
+ });
+
+ it('refuses a record with no pid field at all', () => {
+ const persisted = { odd: persistedSession({ id: 'odd', pid: undefined as unknown as null }) };
+ expect(planRebootRestore(['odd'], persisted, () => true).skipped[0].reason).toBe('not-running');
+ });
+});
+
describe('a workspace that is no longer on disk', () => {
it('is kept out of the offer, so a click cannot scaffold a deleted repo', () => {
const persisted = { gone: persistedSession({ id: 'gone', workingDir: '/tmp/deleted-repo' }) };
From 62ceb4e87b35904b28a039c8102f3dde938b725f Mon Sep 17 00:00:00 2001
From: Michael Grundberg
Date: Thu, 17 Sep 2026 19:49:06 +0200
Subject: [PATCH 18/28] fix(sessions): correct what the missing-pid rule
actually recognises
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
A second real reboot disproved the mechanism the previous commit was built
on. Typing `/exit` does not persist `pid: null`, and the session was
restored anyway.
The pid a session record carries is its `tmux attach-session` process, not
the agent. `/exit` ends the CLI inside the pane, `remain-on-exit` keeps the
pane, and the attach process stays alive throughout — so Codeman's PTY never
exits, no exit handler runs, and the record keeps both its pid and
`status: 'idle'`. The lifecycle log for the session that came back shows
created, started, stale_cleaned and recovered, with no exit event at all,
which is the proof: Codeman never learned the agent was gone.
So nothing durable distinguishes an exited agent from a session that was
idle when the power went, and this pass restores both. Ark0N/Codeman#446 is
about making Codeman notice the dead pane; contrary to what the previous
commit's message claimed, this genuinely does wait on that. Until a record
can say the agent is gone, the user dismisses or closes those sessions.
The rule itself is kept, because a record with no attach process does
describe a session that never started or whose pane died outright, and
refusing it is right. Only its documentation was wrong. The module header,
the branch comment and the test names now say what it recognises instead of
claiming the case it cannot see.
Refs #411
Co-Authored-By: Claude Opus 5 (1M context)
---
src/reboot-restore.ts | 40 ++++++++++++++++++++++---------------
test/reboot-restore.test.ts | 11 +++++-----
2 files changed, 30 insertions(+), 21 deletions(-)
diff --git a/src/reboot-restore.ts b/src/reboot-restore.ts
index d295a34d..85760e8d 100644
--- a/src/reboot-restore.ts
+++ b/src/reboot-restore.ts
@@ -19,12 +19,19 @@
* touching its status, so a pinned session a reboot killed still reads `idle` or
* `busy` and stays eligible.
*
- * Ending the AGENT rather than the session leaves a third shape, and it is the one
- * a real reboot caught this module getting wrong. `/exit` ends the CLI process
- * while the session record survives, and the process-exit handler persists
- * `pid: null` with `status: 'idle'` — indistinguishable by status from a session
- * that was merely idle when the power went. The absent pid is what tells them
- * apart, so a record without one is refused.
+ * ⚠️ Ending the AGENT rather than the session is a shape this module CANNOT
+ * recognise today, and a reboot restores it. `/exit` ends the CLI inside the
+ * pane, `remain-on-exit` keeps the pane, and the PTY Codeman owns is the
+ * `tmux attach-session` process, which stays alive throughout — so no exit
+ * handler runs, no lifecycle `exit` is logged, and the record keeps both its pid
+ * and `status: 'idle'`. Nothing durable distinguishes it from a session that was
+ * simply idle when the power went. Ark0N/Codeman#446 covers making Codeman
+ * notice the dead pane; until a record can say the agent is gone, this pass will
+ * offer those sessions back, and the user dismisses or closes them.
+ *
+ * The `pid` check below is therefore NOT that rule. It refuses a record whose
+ * attach process was already gone, which is a session that never started or
+ * whose pane died outright.
*
* @dependencies types (SessionState), config/cli-registry
* @consumedby web/server (plan build at boot), web/routes/reboot-restore-routes
@@ -172,17 +179,18 @@ export function planRebootRestore(
continue;
}
if (state.pid === null || state.pid === undefined) {
- // The agent had already exited when the machine went down: `/exit` ends the
- // process, and its exit handler persists `pid: null` with `status: 'idle'`
- // before anything else can. Status alone cannot tell that apart from a
- // session that was simply sitting idle when the power went, so without this
- // a reboot restore spawns the agents the user deliberately closed — the
- // exact case the eligibility rule exists to exclude.
+ // No attach process when the record was last written: the session never
+ // started, or its pane died outright rather than its agent exiting inside a
+ // surviving pane. Either way there was nothing running to bring back.
//
- // A heuristic, and deliberately the conservative one. A session that somehow
- // persisted no pid while genuinely running is not offered, and its
- // conversation stays reachable from the Resume list, which is where every
- // session would be without this feature.
+ // ⚠️ This does NOT catch a session the user ended with `/exit`. See the
+ // module header: that leaves the pid in place, because the pid is the tmux
+ // attach process and `remain-on-exit` keeps it alive.
+ //
+ // Conservative on purpose. A session that somehow persisted no pid while
+ // genuinely running is not offered, and its conversation stays reachable
+ // from the Resume list, which is where every session would be without this
+ // feature.
skipped.push({ sessionId, reason: 'not-running' });
continue;
}
diff --git a/test/reboot-restore.test.ts b/test/reboot-restore.test.ts
index 5ed989fd..67bc34e9 100644
--- a/test/reboot-restore.test.ts
+++ b/test/reboot-restore.test.ts
@@ -138,17 +138,18 @@ describe('which dead sessions may be rebuilt', () => {
});
});
-describe('a session whose agent had already exited', () => {
- it('is refused, because `/exit` leaves the record reading idle with no pid', () => {
- // What the process-exit handler persists: the CLI is gone, the record is not,
- // and its status is indistinguishable from a session that was merely idle.
+describe('a session with no attach process in its record', () => {
+ it('is refused, because there was nothing running to bring back', () => {
+ // A session that never started, or whose pane died outright. NOT a session
+ // the user ended with `/exit`: that keeps its pid, because the pid is the
+ // tmux attach process and `remain-on-exit` keeps the pane alive.
const persisted = { exited: persistedSession({ id: 'exited', status: 'idle', pid: null }) };
const plan = planRebootRestore(['exited'], persisted, () => true);
expect(plan.restore).toEqual([]);
expect(plan.skipped).toEqual([{ sessionId: 'exited', reason: 'not-running' }]);
});
- it('still restores the session beside it that was running when the power went', () => {
+ it('still restores the session beside it that was attached when the power went', () => {
const persisted = {
exited: persistedSession({ id: 'exited', pid: null }),
running: persistedSession({ id: 'running', pid: 4242 }),
From 1f61d2129874d0ccbaafb7528963e156ce22c9b4 Mon Sep 17 00:00:00 2001
From: Codeman maintainer
Date: Fri, 18 Sep 2026 13:41:02 +0200
Subject: [PATCH 19/28] docs: correct six stale counts and claims in CLAUDE.md
Each of these was measurable and wrong: the CI note listed 5 excluded
Playwright tests where config/test-suites.ts has 9, never mentioned the
packages/xterm-zerolag-input run that follows the gate, and never
mentioned wiki-sync.yml at all; the format glob note omitted that lint
covers only src/**/*.ts; app.js is ~6.9K lines, not ~6.7K, and
voice-pcm-worklet.js is fetched from JS rather than sitting in the load
order; src/config/ holds 23 files plus the cli-registry/ subdir, not 21,
and nothing said that the repo-root config/ is a different directory;
the route count is ~232 with cases at 34, not ~228 with cases at 30.
Also adds the pointer to docs/wiki/ as the user-facing manual, which the
header describes every other doc surface but not that one.
Co-Authored-By: Claude Opus 5 (1M context)
---
CLAUDE.md | 12 ++++++------
1 file changed, 6 insertions(+), 6 deletions(-)
diff --git a/CLAUDE.md b/CLAUDE.md
index 4ef118e5..7278266d 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -2,7 +2,7 @@
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
-> Deep implementation detail lives in [`docs/architecture-invariants.md`](docs/architecture-invariants.md). This file holds the rules that prevent mistakes; that file holds the mechanisms, file inventories, and the history behind each rule. Pointers below are written as `→ architecture-invariants#anchor`. When the goal is raw throughput, [`docs/SPEEDRUN.md`](docs/SPEEDRUN.md) is the fast-execution protocol (it removes ceremony, never the safety rules here).
+> Deep implementation detail lives in [`docs/architecture-invariants.md`](docs/architecture-invariants.md). This file holds the rules that prevent mistakes; that file holds the mechanisms, file inventories, and the history behind each rule. Pointers below are written as `→ architecture-invariants#anchor`. When the goal is raw throughput, [`docs/SPEEDRUN.md`](docs/SPEEDRUN.md) is the fast-execution protocol (it removes ceremony, never the safety rules here). The user-facing manual is `docs/wiki/` (mirrored to the GitHub wiki by CI; see the CI note under Additional Commands), and `AGENTS.md` deliberately just points here.
>
> **This file is in `.prettierignore` on purpose.** Prettier's markdown printer escapes underscores inside the glob-heavy paths used throughout (`agent-*.jsonl` became `agent-\_.jsonl`, collapsing backtick spans and corrupting a whole paragraph). Do not remove the ignore entry, and do not run `prettier --write` on it.
>
@@ -120,11 +120,11 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
| Dependency doctor | `codeman doctor` (alias `check-deps`; `--json`, `--category core\|office\|other`). Probes Node/Claude CLI/tmux/LibreOffice/MS Office against `config/dependency-registry.ts`; engine is pure given an injectable `ProbeHost` |
| Multi-user accounts | `codeman users add ` / `passwd ` / `list` / `rm ` (writes `~/.codeman/users.json`, mode 0600; see Multi-user mode) |
-**CI**: `.github/workflows/ci.yml` (push to master/main + PRs, Node 22) runs two jobs: **(1)** `check:lockfile`, `typecheck`, `lint`, `check:frontend-syntax`, `format:check`, then a **server boot smoke test** (`tsx src/index.ts web --port 3151` must answer `/api/status` within 30s); **(2)** the **unit/integration test suite** via `npm run test:ci` (`config/vitest.ci.config.ts` — excludes the browser-driven `test/mobile/**` suite, `perf-*` benchmarks, and 5 Playwright tests; globs live in `config/test-suites.ts`). `npm test` runs this same config, so local green == CI green. Tests are tmux-safe in CI: `TmuxManager` no-ops all shell commands under `VITEST` (see Testing).
+**CI**: `.github/workflows/ci.yml` (push to master/main + PRs, Node 22) runs two jobs: **(1)** `check:lockfile`, `typecheck`, `lint`, `check:frontend-syntax`, `format:check`, then a **server boot smoke test** (`tsx src/index.ts web --port 3151` must answer `/api/status` within 30s); **(2)** the **unit/integration test suite** via `npm run test:ci` (`config/vitest.ci.config.ts` — excludes the browser-driven `test/mobile/**` suite, `perf-*` benchmarks, and 9 Playwright tests; globs live in `config/test-suites.ts`), followed by the **`packages/xterm-zerolag-input` package tests** (a bare `npx vitest run` in that directory; its vitest is hoisted by the root `npm ci`, so no separate install, and `npm test` at the root does NOT run them). `npm test` runs this same config, so local green == CI green. Tests are tmux-safe in CI: `TmuxManager` no-ops all shell commands under `VITEST` (see Testing). A third workflow, `wiki-sync.yml`, fires only on master pushes touching `docs/wiki/**` and mirrors that directory to the GitHub wiki (browser edits to the wiki are overwritten by the next sync, so fix pages via `docs/wiki/`).
**Code style**: Prettier (`singleQuote: true`, `printWidth: 120`, `trailingComma: "es5"`) — config lives in the **`"prettier"` key of `package.json`**, not a `.prettierrc` (keeps the repo root short; editors read it natively). `.prettierignore` stays at the root because Prettier resolves it relative to cwd. ESLint flat config (`config/eslint.config.js`) allows `no-console`, warns on `@typescript-eslint/no-explicit-any`. Ignores: `app.js`, `scripts/**/*.mjs`, `src/web/public/vendor/**`, `scripts/remotion/**`.
-**Prettier scope is deliberately narrow.** `npm run format` globs only `src/**/*.ts` and `src/web/public/**`, and `.prettierignore` then exempts most of `src/web/public/*.js` (app.js, styles.css, **mobile.css**, index.html, upload.html, and 15 hand-formatted modules) plus `CLAUDE.md`. Those files are hand-formatted by design; `npm run check:public-assets` and `check:frontend-syntax` are what guard them (NUL bytes + JS syntax), not Prettier. Do not "fix" a file by adding it back to Prettier's scope.
+**Prettier scope is deliberately narrow.** `npm run format` globs only `src/**/*.ts` and `src/web/public/**` (`lint` only `src/**/*.ts`), and `.prettierignore` then exempts most of `src/web/public/*.js` (app.js, styles.css, **mobile.css**, index.html, upload.html, and 15 hand-formatted modules) plus `CLAUDE.md`. Those files are hand-formatted by design; `npm run check:public-assets` and `check:frontend-syntax` are what guard them (NUL bytes + JS syntax), not Prettier. Do not "fix" a file by adding it back to Prettier's scope.
## Common Gotchas
@@ -171,14 +171,14 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
| **Attachments** | `src/attachment-registry.ts`, `attachment-magic`, `generated-artifact-attachments`, `session-attachment-history`, `document-preview-cache`, `document-thumbnailer`, `document-conversion-limiter`, `config/attachment-guard` | See Key Patterns |
| **Plan** | `src/plan-orchestrator.ts`, `src/prompts/*.ts`, `src/templates/` (`claude-md.ts` + `case-template.md`) | `templates/` holds the CLAUDE.md scaffold generated into new cases |
| **Web** | `src/web/server.ts` ★, `sse-events.ts`, `routes/*.ts` (25 modules + barrel; `session-routes.ts` ★), `route-helpers.ts`, `ports/*.ts`, `middleware/auth.ts`, `schemas.ts`, `self-update.ts`, `plan-usage-latest.ts`, `ws-connection-registry.ts`, `heic-jpeg-converter.ts` + `heic-jpeg-worker.ts` | |
-| **Frontend** | `src/web/public/app.js` (~6.7K lines, core) + 32 modules + `sw.js` | See Frontend section for the load order, which is authoritative |
+| **Frontend** | `src/web/public/app.js` (~6.9K lines, core) + 32 modules + `sw.js` (+ `voice-pcm-worklet.js`, fetched from JS, not in the load order) | See Frontend section for the load order, which is authoritative |
| **Types** | `src/types/index.ts` (barrel) → 22 domain files; also `src/types.ts` root re-export | See `@fileoverview` in index.ts |
★ = Large, central file (>50KB) — read its `@fileoverview` first. All files have `@fileoverview` JSDoc — read that before diving in. Discovery aid: `grep -l '@fileoverview' src/web/routes/*.ts` lists all route modules; same grep works for `src/types/`, `src/web/public/*.js`.
**Local packages**: `packages/xterm-zerolag-input/` (local echo overlay, single-source, see Gotchas). `packages/gesture-control/` (`codeman-gesture-control`, hand-tracking overlay source, built via `npm run build:gesture`).
-**Config**: `src/config/` — 21 files, no barrel (`index.ts`) exists; import from the specific file.
+**Config**: `src/config/` — 23 files plus the `cli-registry/` subdir, no barrel (`index.ts`) exists; import from the specific file. ⚠️ There are TWO `config/` directories: the repo-root `config/` holds tooling only (ESLint, knip, the vitest configs, `test-suites.ts`), while runtime config lives in `src/config/`. Throughout this file a bare `config/.ts` in a code context means `src/config/.ts`.
**Utilities**: `src/utils/` — re-exported via index. Key: `CleanupManager`, `LRUMap` (⚠ NOT in the barrel — import from `./utils/lru-map.js` directly), `StaleExpirationMap`, `BufferAccumulator`, `stripAnsi`, `Debouncer`, `KeyedDebouncer`. Also: `claude-cli-resolver`/`opencode-cli-resolver`/`codex-cli-resolver`/`gemini-cli-resolver`/`antigravity-cli-resolver`/`pi-cli-resolver`/`grok-cli-resolver`/`deepseek-cli-resolver`/`omp-cli-resolver` (CLI path resolution, one per `SessionMode`, all nine sharing the lookup chain in `cli-executable-resolver`: server PATH, then that CLI's install dirs, then an interactive login shell LAST, since it is the only step that spawns anything and it is what finds nvm/Homebrew installs under a service manager's minimal PATH; ⚠ `pi-`, `grok-` and `deepseek-cli-resolver` additionally probe the binary's identity, since `pi` is a generic name, `grok` has npm squatters, and Debian ships an unrelated `dsh`), `file-query` (⚠ Files-panel search matcher, glob-by-two-pointer, never RegExp), `string-similarity` (fuzzy matching), `regex-patterns` (ANSI/token/spinner patterns), `assertNever` (exhaustive checks), `token-validation` (auth tokens), `nice-wrapper` (process priority), `shell-resolver` (⚠ resolves a real login shell for `mode: 'shell'`; the literal string `$SHELL` used to be expanded by the SERVER's shell, which is empty in a container), `event-loop-monitor` (a sync `execSync` freezes the port while the process stays alive, leaving no trace), `dependency-checker` + `dependency-report` (the `codeman doctor` probe engine, registry in `config/dependency-registry.ts`).
@@ -381,7 +381,7 @@ Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. L
### API Routes
-~228 handlers across 25 route files in `src/web/routes/`: system (56), sessions (34), cases (30), files (17), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), approvals (4), readmymind (4), me (2), teams (2), tab-layout (2), search (1), hooks (1), clipboard (1), status-telemetry (1), voice (1 + the `/ws/voice/stream` relay), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details.
+~232 handlers across 25 route files in `src/web/routes/`: system (56), sessions (34), cases (34), files (17), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), approvals (4), readmymind (4), me (2), teams (2), tab-layout (2), search (1), hooks (1), clipboard (1), status-telemetry (1), voice (1 + the `/ws/voice/stream` relay), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details.
**HTTP contract** (stable since 0.9.x, see `docs/versioning-policy.md`; full envelope/status/error-code/SSE spec in `docs/api-reference.md`): responses use the `ApiResponse` envelope — `{ success: true, data? }` or `{ success: false, error, errorCode }` (`src/types/api.ts`). `/api/v1/*` is a versioned alias of `/api/*` (URL rewrite in `server.ts`).
From dee674d3e2fc93553149c4c84f868deecb226e97 Mon Sep 17 00:00:00 2001
From: Codeman maintainer
Date: Fri, 18 Sep 2026 13:41:45 +0200
Subject: [PATCH 20/28] fix(install): point the launcher-only caveat at the
thing that resolves it
The caveat #429 added ends with "see the docs above", and "the docs
above" is CLI_DOCS[$i], which for DeepSeek is the upstream harness repo.
Per docs/deepseek-integration.md the harness ships only the web,
headless and base profiles, so following that link and running
`npm install -g @deepseek-ai/dsh` leaves the reader exactly where the
caveat is warning them about: a dsh that cannot drive a pane. What
actually resolves it is Codeman's own Run dropdown, which offers
"DeepSeek: add a terminal profile..." and installs one in a click.
The new wording stays generic for any future launcherProfile entry,
since Codeman is the thing being installed at all three call sites.
Also flips one word in the generator: the comment said "see
installCommandFor below" and that function is defined above it.
Co-Authored-By: Claude Opus 5 (1M context)
---
install.sh | 2 +-
scripts/generate-cli-catalog.mts | 2 +-
2 files changed, 2 insertions(+), 2 deletions(-)
diff --git a/install.sh b/install.sh
index 8b5672bc..29b6a13d 100755
--- a/install.sh
+++ b/install.sh
@@ -587,7 +587,7 @@ cli_catalog_print_install_hints() {
elif [[ -n "${CLI_DOCS[$i]}" ]]; then
echo -e " ${CLI_LABELS[$i]}: see ${CYAN}${CLI_DOCS[$i]}${NC}"
if [[ "${CLI_LAUNCHER_ONLY[$i]}" == "1" ]]; then
- echo -e " (its package installs a launcher only — it needs a profile that can drive a pane, see the docs above)"
+ echo -e " (installs a launcher only: it still needs a terminal profile, and Codeman's Run menu can add one)"
fi
fi
done
diff --git a/scripts/generate-cli-catalog.mts b/scripts/generate-cli-catalog.mts
index e17df0ec..de0d7934 100644
--- a/scripts/generate-cli-catalog.mts
+++ b/scripts/generate-cli-catalog.mts
@@ -151,7 +151,7 @@ export function renderInstallShBlock(entries: CliEntry[] = STOCK_CLIS): string {
labels.push(shQuote(entry.label));
enabled.push(entry.enabled ? '1' : '0');
// Parallel to CLI_IDS: 1 when this entry's install command installs a launcher rather
- // than something that can drive a pane on its own (see installCommandFor below). Purely
+ // than something that can drive a pane on its own (see installCommandFor above). Purely
// derived from discovery.launcherProfile — install.sh's hint printer reads this to add a
// caveat instead of hardcoding which id it means.
launcherOnly.push(entry.discovery.launcherProfile ? '1' : '0');
From ea5323d9908fd9ffe5651841076684627435c34a Mon Sep 17 00:00:00 2001
From: Codeman maintainer
Date: Fri, 18 Sep 2026 13:42:44 +0200
Subject: [PATCH 21/28] test(input): pin the batched commit-plus-Enter ordering
#441 fixes
The unit harness proves WHICH candidate gets forwarded; the ordering is
the half that shipped the bug, and only a real xterm shows it. The new
browser case dispatches the character's keydown, its composed insertText
and Enter's keydown in ONE page task, the shape an Android soft keyboard
delivers through a single InputConnection transaction, and asserts what
reaches the send path.
Verified in both directions on this machine: with the drain in place the
wire is `o\r`; with the drain removed (master's behaviour) it is `\r` and
the character is gone entirely, because by the time the zero-delay timer
runs xterm has emitted the `\r` and bumped the canonical counter past the
candidate's snapshot, so the candidate stands down. The other four cases
pass in both states.
CLAUDE.md now names the decision point, what it costs (a keydown decides
with less evidence than the timer did) and why that is safe for Enter,
and says that the pin lives in a suite the CI gate does not run.
Co-Authored-By: Claude Opus 5 (1M context)
---
CLAUDE.md | 2 +-
...rminal-keycode229-recovery.browser.test.ts | 83 ++++++++++++++++++-
2 files changed, 83 insertions(+), 2 deletions(-)
diff --git a/CLAUDE.md b/CLAUDE.md
index 7278266d..4690fcab 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -298,7 +298,7 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
### Frontend
-Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. Load order: `constants.js`(1) → `i18n.js`(1.5) → `mobile-handlers.js`(2) → `voice-input.js`(3) → `notification-manager.js`(4) → `keyboard-accessory.js`(5) → `input-cjk.js`(5.5) → `terminal-keycode229-recovery.js`(5.55) → `sanitize-html.js`(5.6) → `app.js`(6) → `tab-rail-resize.js`(6.5) → `terminal-ui.js`(7) → `respawn-ui.js`(8) → `ralph-panel.js`(9) → `orchestrator-panel.js`(9.5) → `cron-ui.js`(9.7) → `settings-ui.js`(10) → `panels-ui.js`(11) → `readmymind-ui.js`(11.3) → `ultracode-panel.js`(11.5) → `approvals-ui.js`(11.6) → `admin-ui.js`(11.7) → `session-ui.js`(12) → `webview-tabs.js`(12.5) → `mobile-overview.js`(12.55) → `home-sessions.js`(12.56) → `entrance-animations.js`(12.6) → `ralph-wizard.js`(13) → `api-client.js`(14) → `subagent-windows.js`(15) → `ultracode-windows.js`(15.5) → `session-lineage.js`(15.6) → `image-input.js`(16). `i18n.js` translates static + newly inserted application DOM while skipping terminal/response/file/user-name surfaces; `input-cjk.js` handles CJK IME composition via an always-visible textarea below the terminal (`window.cjkActive` blocks xterm's onData). `terminal-keycode229-recovery.js` forwards a committed `input` event that xterm's `_inputEvent` guard drops (Chrome-on-Android soft keyboards send `composed: true` after a keydown), and only when xterm emitted no canonical data for that keystroke.
+Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. Load order: `constants.js`(1) → `i18n.js`(1.5) → `mobile-handlers.js`(2) → `voice-input.js`(3) → `notification-manager.js`(4) → `keyboard-accessory.js`(5) → `input-cjk.js`(5.5) → `terminal-keycode229-recovery.js`(5.55) → `sanitize-html.js`(5.6) → `app.js`(6) → `tab-rail-resize.js`(6.5) → `terminal-ui.js`(7) → `respawn-ui.js`(8) → `ralph-panel.js`(9) → `orchestrator-panel.js`(9.5) → `cron-ui.js`(9.7) → `settings-ui.js`(10) → `panels-ui.js`(11) → `readmymind-ui.js`(11.3) → `ultracode-panel.js`(11.5) → `approvals-ui.js`(11.6) → `admin-ui.js`(11.7) → `session-ui.js`(12) → `webview-tabs.js`(12.5) → `mobile-overview.js`(12.55) → `home-sessions.js`(12.56) → `entrance-animations.js`(12.6) → `ralph-wizard.js`(13) → `api-client.js`(14) → `subagent-windows.js`(15) → `ultracode-windows.js`(15.5) → `session-lineage.js`(15.6) → `image-input.js`(16). `i18n.js` translates static + newly inserted application DOM while skipping terminal/response/file/user-name surfaces; `input-cjk.js` handles CJK IME composition via an always-visible textarea below the terminal (`window.cjkActive` blocks xterm's onData). `terminal-keycode229-recovery.js` forwards a committed `input` event that xterm's `_inputEvent` guard drops (Chrome-on-Android soft keyboards send `composed: true` after a keydown), and only when xterm emitted no canonical data for that keystroke. ⚠️ **That decision is settled at the NEXT keydown as well as on its own zero-delay timer** (#441): the drain runs from xterm's custom key handler, which fires BEFORE xterm processes that key, so a soft keyboard that commits the last character and sends Enter in one InputConnection transaction puts the character on the wire ahead of the `\r`. On the timer alone that character is not merely late, it is LOST: xterm emits the `\r` first and bumps the canonical counter past the candidate's snapshot, so the candidate stands down (measured, `hell\r` where the user typed `hello`). The trade is that a keydown decides with less evidence than the timer did, since xterm's own keyCode-229 rescue has not run yet; that is safe for Enter, which clears the textarea so the pending diff emits nothing. Ordering is pinned by `test/terminal-keycode229-recovery.browser.test.ts`, which the CI gate does NOT run.
**Entrance animations** (`entrance-animations.js`, all OFF by default): opt-in animations for the four things that appear when work starts, chosen per surface via `data-tab-anim` / `data-term-anim` / `data-win-anim` / `data-line-anim` on ``. Defaults are the `legacy` theme, so an untouched install behaves exactly as before and every hook short-circuits on its first line. ⚠️ Tabs and connection lines are **destroyed mid-animation** on every re-render (`_fullRenderSessionTabs()` replaces the strip's innerHTML; `_updateConnectionLinesImmediate()` does `svg.innerHTML = ''`), so both are tracked by id and re-applied to the fresh element with a **negative `animation-delay`** to resume rather than restart. ⚠️ The terminal-pane styles may animate **transform / opacity / clip-path only**, xterm's FitAddon derives rows+cols from `getComputedStyle(parent).width/height`, so animating width/height/padding there would resize the PTY; `test/entrance-animations.test.ts` pins that property allowlist, plus the rule→keyframes→theme-option chain a style silently does nothing without. ⚠️ **`blur` is the ONE style that puts a `filter` on the terminal container**, against the standing rule, because every alternative was measured against a live xterm and does not work: a `backdrop-filter` veil on `::before` blurs perfectly while STATIC and Chrome silently drops the backdrop the moment ANY animation runs on that pseudo-element (the veil computes `blur(15.3px)` and the text behind it stays razor sharp), and driving the radius from rAF buys the same full-screen blur per frame plus main-thread work. The cost the rule exists to avoid is inherent to blurring a terminal, so the style buys it knowingly: opt-in, OFF by default, one ~520ms run per session open, class straight back off, `will-change` still unset. Worst-case price, headless SwiftShader with no GPU: frame deltas 16.7ms → 33.3ms for the run, against 16.7ms flat for `fade`. Do not generalise it — a second filtered terminal style needs its own measurement. ⚠️ The `blur` connection line animates `filter` too, so both kinds of line hold their glow in **`--line-glow`** and both of its keyframes say `blur(N) var(--line-glow)`: the function lists then match and interpolate, instead of the glow vanishing for the run and popping back (a lineage line's glow is a different colour entirely, set per element). Its 100% frame deliberately omits `opacity` so the endpoint comes from the element's own resting value — 0.9 subagent, 0.72 lineage, 0.95 working — which is what `line-enter-fade`'s hardcoded 0.9 gets wrong. ⚠️ Window styles other than `beam` transform the window, which moves the rect its connection line is aimed at; `beam` deliberately animates opacity/filter only so its line can draw toward a stable target. Persisted to its own `codeman:*Anim` localStorage keys (per-device, deliberately NOT in the `.strict()` `SettingsUpdateSchema`); picker in App Settings → Appearance, full per-surface lab at `?animlab=1`.
diff --git a/test/terminal-keycode229-recovery.browser.test.ts b/test/terminal-keycode229-recovery.browser.test.ts
index 8f980fc0..18b08d1e 100644
--- a/test/terminal-keycode229-recovery.browser.test.ts
+++ b/test/terminal-keycode229-recovery.browser.test.ts
@@ -9,7 +9,10 @@
* (stopPropagation, not stopImmediatePropagation) does not silence it;
* - a `composed: true` insertText preceded by a keydown — the shape Chrome on
* Android delivers — is dropped by xterm and recovered by us, exactly once;
- * - a keystroke xterm DOES handle is delivered exactly once, not twice.
+ * - a keystroke xterm DOES handle is delivered exactly once, not twice;
+ * - a character committed in the SAME page task as Enter reaches the send
+ * path ahead of the `\r`, which is the ordering the zero-delay timer
+ * alone cannot produce.
*
* Browser-driven, so it is excluded from `npm run test:ci` like the other
* Playwright suites. Run locally:
@@ -146,6 +149,84 @@ describe('orphaned terminal input recovery wiring', () => {
expect(second.sent.join('')).toBe('z');
});
+ /**
+ * The batched shape an Android soft keyboard actually delivers when the user
+ * taps the last character and then Enter: the character's keydown, its
+ * `composed: true` insertText, and Enter's keydown all land in ONE page task,
+ * before any zero-delay timer can run.
+ *
+ * This is the ordering half of the fix, and the half the unit harness cannot
+ * reach: the unit tests prove WHICH candidate is forwarded, this proves WHEN.
+ * Resolving the pending candidate only on its 0 ms timer loses the character
+ * outright here, because by the time that timer runs xterm has already
+ * emitted the `\r` and bumped the canonical counter past the candidate's
+ * snapshot, so it stands down. Draining at the next keydown, from xterm's
+ * custom key handler (which runs before xterm processes that key), puts the
+ * character on the wire ahead of the `\r`.
+ */
+ async function batchedCommitThenEnter(data: string) {
+ return page.evaluate(async (text) => {
+ const app = (window as any).app;
+ const textarea = document.querySelector('.xterm-helper-textarea') as HTMLTextAreaElement;
+ const originalSessionId = app.activeSessionId;
+ const originalLocalEcho = app._localEchoEnabled;
+ const originalSendInput = app._sendInputAsync;
+ const originalPendingInput = app._pendingInput;
+ const originalLastKeystrokeTime = app._lastKeystrokeTime;
+ const sent: string[] = [];
+
+ try {
+ app.activeSessionId = 'cod388-browser-batched';
+ app._localEchoEnabled = false;
+ app._pendingInput = '';
+ app._lastKeystrokeTime = 0;
+ app._sendInputAsync = (_sessionId: string, chunk: string) => sent.push(chunk);
+ textarea.focus();
+
+ // One task, no awaits between the three dispatches.
+ const charDown = new KeyboardEvent('keydown', {
+ key: 'Unidentified',
+ bubbles: true,
+ cancelable: true,
+ composed: true,
+ });
+ Object.defineProperties(charDown, { keyCode: { value: 65 }, which: { value: 65 } });
+ textarea.dispatchEvent(charDown);
+
+ textarea.value = text;
+ textarea.dispatchEvent(
+ new InputEvent('input', { data: text, inputType: 'insertText', bubbles: true, composed: true })
+ );
+
+ const enterDown = new KeyboardEvent('keydown', {
+ key: 'Enter',
+ code: 'Enter',
+ bubbles: true,
+ cancelable: true,
+ composed: true,
+ });
+ Object.defineProperties(enterDown, { keyCode: { value: 13 }, which: { value: 13 } });
+ textarea.dispatchEvent(enterDown);
+
+ await new Promise((resolve) => setTimeout(resolve, 80));
+ return { wire: sent.join('') };
+ } finally {
+ app.activeSessionId = originalSessionId;
+ app._localEchoEnabled = originalLocalEcho;
+ app._sendInputAsync = originalSendInput;
+ app._pendingInput = originalPendingInput;
+ app._lastKeystrokeTime = originalLastKeystrokeTime;
+ textarea.value = '';
+ }
+ }, data);
+ }
+
+ it('delivers a character committed in the same task as Enter BEFORE the carriage return', async () => {
+ const { wire } = await batchedCommitThenEnter('o');
+ // Not '\r' (character lost, the defect) and not '\ro' (recovered too late).
+ expect(wire).toBe('o\r');
+ });
+
it('sends nothing for a keydown that produces no input event', async () => {
const { sent } = await keystroke({ data: 'q', dispatchInput: false, keyCode: 65 });
expect(sent).toEqual([]);
From bb8ada7e5fc5a3991db4392f972e99b4c8d53c12 Mon Sep 17 00:00:00 2001
From: Codeman maintainer
Date: Fri, 18 Sep 2026 13:46:04 +0200
Subject: [PATCH 22/28] fix(reboot-restore): the merge-time items from the #442
review
Seven things, none of which changes what the feature does.
1. The rebuilt Session dropped `nameSource`, so the constructor re-inferred
it from the name: a session the user renamed by hand to something shaped
like `w-` came back as `placeholder`, and with auto-naming on the
next prompt overwrote their name. The route persists right after, so the
loss went to disk. `restoreMuxSessions()` already passes it.
2. The already-live sets were snapshotted once before a loop that awaits a
real `startInteractive()` per entry, so by the tenth entry the snapshot
was tens of seconds old and a conversation resumed by hand from the
Resume list in that window was invisible to it: two panes on one
transcript, the exact thing the check exists to prevent. Both sets are
now read per iteration, and the late case is spent rather than re-offered
for the same reason the batch case is.
3. Auto-resume no longer re-arms the pre-reboot `autoResumeAt` on this path.
The stamp predates the reboot and the pane is new, so honouring it meant
one click had every restored session type `continue` into itself about a
minute later, unattended, against the route header's own promise that a
restored session comes back idle and disarmed. The setting stays ENABLED,
so it re-arms on the next real limit message. A Codeman restart still
re-arms from the stamp, because the limit footer will not reprint on its
own; the new option exists only to tell the two paths apart.
4. `discardPartiallyBuiltSession()` now also calls `recordSessionStopped()`
and `ralphTracker.fullReset()`, the two teardown steps `_doCleanupSession`
performs that it was missing. Cosmetic, but a run left open reads as
still going in the away digest.
5. A restored claude session gets `seedAgentSessionPreamble()` like both
create paths, so the agent skill's bootstrap stays a two-line loader.
6. The heuristic's container comment was wrong in one direction and quiet
about the real gap: after a genuine host reboot a containerized Codeman
sees the host's short uptime and the banner does appear. What it cannot
see is a container-only restart, which is where this would help most.
7. The banner is hidden in a solo window, which shows one session and has
no tab strip to put restored ones in.
Also reverts 17 of the 18 hunks in docs/api-reference.md, which were
Prettier reformatting of prose the PR does not otherwise touch (docs/ is
outside the format glob), keeping only the Reboot restore section and
repairing the two continuation lines that reformat de-indented; renumbers
reboot-restore-ui.js to @loadorder 11.65, since 11.7 is admin-ui.js, which
loads after it; and gives the feature its CLAUDE.md entry plus a route
test for the multi-user workspace-forbidden branch, the only new rule that
had nothing behind it.
Co-Authored-By: Claude Opus 5 (1M context)
---
CLAUDE.md | 10 +-
docs/api-reference.md | 168 +++++++++++-----------
src/reboot-restore.ts | 16 ++-
src/web/ports/session-port.ts | 14 +-
src/web/public/reboot-restore-ui.js | 2 +-
src/web/public/styles.css | 4 +
src/web/routes/reboot-restore-routes.ts | 54 ++++++-
src/web/server.ts | 21 ++-
test/routes/reboot-restore-routes.test.ts | 48 ++++++-
9 files changed, 228 insertions(+), 109 deletions(-)
diff --git a/CLAUDE.md b/CLAUDE.md
index 4690fcab..b3d765ac 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -170,8 +170,8 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
| **Search** | `src/search-service.ts` | Pure in-memory core for `GET /api/search` |
| **Attachments** | `src/attachment-registry.ts`, `attachment-magic`, `generated-artifact-attachments`, `session-attachment-history`, `document-preview-cache`, `document-thumbnailer`, `document-conversion-limiter`, `config/attachment-guard` | See Key Patterns |
| **Plan** | `src/plan-orchestrator.ts`, `src/prompts/*.ts`, `src/templates/` (`claude-md.ts` + `case-template.md`) | `templates/` holds the CLAUDE.md scaffold generated into new cases |
-| **Web** | `src/web/server.ts` ★, `sse-events.ts`, `routes/*.ts` (25 modules + barrel; `session-routes.ts` ★), `route-helpers.ts`, `ports/*.ts`, `middleware/auth.ts`, `schemas.ts`, `self-update.ts`, `plan-usage-latest.ts`, `ws-connection-registry.ts`, `heic-jpeg-converter.ts` + `heic-jpeg-worker.ts` | |
-| **Frontend** | `src/web/public/app.js` (~6.9K lines, core) + 32 modules + `sw.js` (+ `voice-pcm-worklet.js`, fetched from JS, not in the load order) | See Frontend section for the load order, which is authoritative |
+| **Web** | `src/web/server.ts` ★, `sse-events.ts`, `routes/*.ts` (27 modules + barrel; `session-routes.ts` ★), `route-helpers.ts`, `ports/*.ts`, `middleware/auth.ts`, `schemas.ts`, `self-update.ts`, `plan-usage-latest.ts`, `ws-connection-registry.ts`, `heic-jpeg-converter.ts` + `heic-jpeg-worker.ts` | |
+| **Frontend** | `src/web/public/app.js` (~6.9K lines, core) + 33 modules + `sw.js` (+ `voice-pcm-worklet.js`, fetched from JS, not in the load order) | See Frontend section for the load order, which is authoritative |
| **Types** | `src/types/index.ts` (barrel) → 22 domain files; also `src/types.ts` root re-export | See `@fileoverview` in index.ts |
★ = Large, central file (>50KB) — read its `@fileoverview` first. All files have `@fileoverview` JSDoc — read that before diving in. Discovery aid: `grep -l '@fileoverview' src/web/routes/*.ts` lists all route modules; same grep works for `src/types/`, `src/web/public/*.js`.
@@ -243,6 +243,8 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
**Hook events**: Claude Code hooks trigger via `/api/hook-event`. Key events: `permission_prompt`, `elicitation_dialog`, `elicitation_complete`, `elicitation_response`, `idle_prompt`, `stop`, `teammate_idle`, `task_completed`, `prompt_submitted` (UserPromptSubmit, #367: a Claude pane reports its live conversation id first-hand). See `src/hooks-config.ts`; upstream hook semantics mirrored in `docs/claude-code-hooks-reference.md`. ⚠️ **Every claude session INSTALLS the hooks block into its workspace** (`applyWorkspaceHooks` in hooks-config.ts → `ensureCodemanHooks`, an add-only merge that keeps a user's own handlers), from EVERY claude create path — both interactive routes, cron fires, legacy scheduled runs, the plan-orchestrator one-shots — and from `restoreMuxSessions()` for sessions recovered on server start (that boot sweep skips a workspace that no longer exists, so a deleted repo with a surviving tmux session is never resurrected as an empty dir). Before 2026-08-15 hooks were written ONLY when Codeman created the case DIRECTORY, so a linked case / cloned repo — where most sessions actually run — had no hooks at all and every hook-driven surface was silently dead there: an AskUserQuestion dialog blocked the pane while the tab and the phone overview both read a calm `idle`, with no Approvals Inbox item, no push, no definitive `stop`/`idle_prompt` for respawn and no `stop`/`blocked` for the wait endpoints. The escape hatch is the synced `workspaceHooksEnabled` setting (App Settings → Agents & CLIs → Claude, **default ON**); OFF restores the old behavior, where a Codeman block that is already there is still refreshed when stale (COD-91) but one is never added. ⚠️ Route the decision through `applyWorkspaceHooks` rather than calling `ensureCodemanHooks` at a new site, or the setting silently stops applying to that path. ⚠️ Claude Code RE-READS `settings.local.json`, so an already-running session starts firing hooks without a restart (measured 2026-08-15) — and the notification for a blocking dialog is delayed by Claude Code (~30s), so the alert trails the dialog. ⚠️ An AskUserQuestion / plan-selection dialog arrives as **`permission_prompt`**, not `elicitation_dialog` (that one is MCP elicitation), so it renders as the RED "needs you" alert, not the yellow idle one.
+**Reboot restore** (#411/#442, `src/reboot-restore.ts` pure + `web/reboot-restore-registry.ts` + `routes/reboot-restore-routes.ts` + `reboot-restore-ui.js`): a host reboot takes the tmux server with it, so every pane dies and the board comes up empty with no explanation. At boot Codeman works out which sessions that reboot destroyed, holds the plan IN MEMORY (no new state file, and a server restart simply drops the offer), and the banner asks. ⚠️ **The heuristic decides whether to ASK, never whether to act**: two signals have to agree (the socket holds no panes at all while state still lists sessions, AND the host booted after the newest persisted activity), and a wrong yes costs one dismissable line rather than N CLI processes nobody asked for. ⚠️ Rebuilding is TAKE-then-build: entries leave the plan synchronously before the first `await` and the route is single-flighted per owner, so a double-click or a second device cannot put two panes on one conversation. Anything that never became a pane goes BACK on offer, with one deliberate exception, `already-live`, which unlike a missing workspace or a withdrawn grant cannot stop being true. ⚠️ Three things are re-checked at click time rather than trusted from boot (the owner's privilege grant, the workspace still being on disk, and the conversation not already being live), and the already-live sets are read FRESH per iteration rather than snapshotted: the loop awaits a real `startInteractive()` per entry, so a snapshot taken before it is tens of seconds stale by the tenth entry and would miss a conversation the user resumed by hand in that window. The confinement re-check is keyed on the entry's OWNER, never the caller, or an admin spending another user's entry is waved through by `isWorkingDirAllowed`. ⚠️ A rebuilt session comes back **attached, idle and disarmed**: the pane is NEW, so terminal scrollback is gone (the banner says so) while the conversation continues, respawn controllers and Ralph loops are never re-armed, and `reapplyPersistedSessionState(..., { rearmAutoResumeSchedule: false })` keeps auto-resume ENABLED but drops the pre-reboot `autoResumeAt` stamp, or one click has every restored session type `continue` into itself a minute later, unattended. That option exists only for this path; a Codeman restart still re-arms, because the limit footer will not reprint on its own. ⚠️ The rebuild passes `nameSource` through, or the constructor re-infers it from the name and a hand-renamed session shaped like `w-` comes back as `placeholder` for auto-naming to overwrite. ⚠️ Claude-mode only (others carry their conversation id in their own config object), and remote/docker sessions are never offered (`remote-or-docker`), because both need another host or container to be up. ⚠️ A failed rebuild is undone with `discardPartiallyBuiltSession()`, deliberately NOT `cleanupSession()`: the delete path would count the session's tokens into the lifetime totals, demote a pinned record to `stopped` (which this pass reads as an intentional kill, making the session permanently unrestorable) and recursively remove the WORKSPACE's `.claude-images`. ⚠️ `os.uptime()` reports the HOST's uptime, which a container shares, and that cuts both ways: after a genuine host reboot a containerized Codeman does see a short uptime and the banner works, but a container-only restart is invisible to it, which is the case where this would help most. Tests: `test/reboot-restore.test.ts`, `test/routes/reboot-restore-routes.test.ts`, `test/routes/reboot-restore-rebuild-failure.test.ts`, `test/discard-partially-built-session.test.ts`.
+
**Approvals Inbox** (cross-session queue of prompts waiting on a human; `approvalsInboxEnabled`, SYNCED, default OFF: every surface is opt-in; only the store and answer endpoints run regardless, so flipping it ON shows anything already pending): `web/approval-inbox.ts` is a `sessionWaits`-style singleton fed by `/api/hook-event`, holding at most ONE item per session (a new prompt supersedes), claude-mode only, in-memory. Cards are answered via `POST /api/approvals/:id/answer`, which sends a digit / Esc / idle-prompt text through `writeViaMux` (menu answers never carry `\r`). ⚠️ `option` digits are accepted ONLY when they match options parsed from the captured pane frame, and the answer path RE-CAPTURES the pane first (a dialog that no longer parses on screen means the keystroke would land in the composer, so refuse with 409). ⚠️ Resolution on the heuristic `working` signal ALONE is restricted to `idle` items; a permission/question item gets the pane-VERIFIED variant on that same signal (`resolveIfDialogGone()` → `verifyStillAnswerable()`), so the heuristic only decides when to LOOK and the screen decides the outcome. That is what clears a dialog answered in the terminal mid-turn; the other definitive signals are `stop`, `elicitation_complete`/`elicitation_response`, exit/delete, answer, supersede and the 12h TTL. ⚠️ **Viewing a session ACKNOWLEDGES its idle item, it does not resolve it** (`POST /api/approvals/session/:sessionId/viewed` → `acknowledgedAt` → `approval:updated`): the item stays pending (still answerable, still Read My Mind context) and only stops arming the yellow tab alert. That flag is what makes the clear durable, since the view-clears-idle rule used to live in one browser's memory and `seedApprovals()` re-armed the alert on the next reload while other devices never heard about it at all; the local half is `markIdleAlertSeen()` (app.js), called from BOTH `selectSession` paths, including the already-active early return, where a click could otherwise never clear the alert. ⚠️ **Only a HUMAN opening a session acknowledges**: `selectSession(id, { auto: true })` marks the three selections the APP makes (boot restore, a solo window opening its target, the fallback after the active session is closed) and skips the acknowledgement, so a page load cannot silently spend an alert the user never saw. The flag defaults to user-initiated, so an untagged call site fails toward acknowledging rather than toward an alert nothing can clear; `test/session-select-ack-gate.test.ts` pins both the gate and the tagged call sites. Idle-only by construction (`acknowledge()` defaults to `['idle']`): looking at a permission/question dialog does not answer it. ⚠️ Same rule on the input path: `_ackDelivery` (app.js) spends the IDLE alert only, via that same `markIdleAlertSeen()`. It used to `clearPendingHooks(sessionId)` with no kind, so one keystroke wiped a RED alert on that device while the dialog was still up, the other devices stayed red, and a reload re-seeded it. ⚠️ Claude Code fires no "permission answered" hook (only `elicitation_complete`/`elicitation_response`, i.e. the question flavor), so an answered-in-the-terminal dialog would otherwise sit pending until `stop`: `GET /api/approvals` therefore runs a **staleness sweep** over the caller's own items via `verifyStillAnswerable()`, which is deliberately the conservative check the answer path uses (only an item whose ORIGINAL frame parsed options can be dropped, so an unreadable capture keeps the alert rather than losing a live one). ⚠️ **`applyCapture()` is therefore ADD-ONLY for `options`**: a re-capture that parses nothing must never erase a parse an earlier one found. Claude Code delays the Notification hook behind the dialog (measured 6s, documented ~30s), so the 600ms re-capture routinely lands on a frame the user has ALREADY answered; clearing the field there made the item permanently unsweepable, because `verifyStillAnswerable()` reads a MISSING `options` as "we never could read this dialog" and keeps such items answerable by design. The red "needs you" then survived every sweep AND every page reload, went away only on `stop` (2026-08-20: a confirmed question left a tab flowing red for ~8 minutes while the turn ran on), and the stale card still accepted an answer, typing a bare `1` into a composer with no dialog under it. Pinned by `test/approval-inbox.test.ts`. ⚠️ A frame that parses no options is CONCLUSIVE in exactly two cases, and the second one closes the late-hook hole: the item once parsed options (they cannot vanish while the dialog is up), or the frame shows Claude actively running a turn. A modal dialog BLOCKS the turn, so the two cannot coexist — measured on v2.1.237, a live-dialog frame carries neither the `… (13s` timer NOR the `esc to interrupt` footer, which the dialog replaces with `Enter to select · ↑/↓ to navigate · Esc to cancel`. Anything else stays answerable, so an unreadable capture still keeps the alert. That second signal is reached by a delayed staleness pass (`STALE_CHECK_DELAY_MS`, 3s) scheduled alongside the re-capture, because a prompt answered BEFORE the hook lands creates an item whose FIRST capture already has no dialog in it: nothing ever parsed, `stop` may have fired already, and the alert then outlived reloads until the 12h TTL. ⚠️ That pass must stay comfortably LATER than `RECAPTURE_DELAY_MS`, whose whole reason for existing is that the hook can beat Ink to the screen — resolving inside the paint window would clear the alert for a dialog that was about to appear. The frontend seeds from `GET /api/approvals` in `handleInit` **regardless of the setting**: the seed re-arms the tab-alert state machine (`setPendingHook`) unconditionally, and only populating `this.approvals` (the inbox surfaces) is gated — seeding used to be gated wholesale, which left a reloaded page with NO red tab while a permission dialog sat blocking a session (2026-08-15); `_onApprovalResolved` clears the pending-hook alert unconditionally for the same reason. ⚠️ The red/yellow tab alert itself is a STEADY border/background/dot with a pulse on top: the original keyframes swung to transparent at 0%/100%, so half of every cycle looked like a normal tab. Push Approve/Deny buttons stay gated on the setting (`sendPushNotifications` strips `actions`/`approvalId` when OFF) and are answered from `sw.js` directly so they work with no tab open. Surfaces (all gated on the setting): header bell (marker-hidden until count > 0, phones never show it) + drawer (`approvals-ui.js`), phone overview NEEDS YOU answer strips (`mobile-overview.js`). Design: `docs/approvals-inbox-plan.md`.
**Read My Mind intent profiles** (phase 1 of `docs/readmymind-plan.md`; `readMyMindEnabled`, SYNCED, default OFF): per-CASE profiles (user-stated `goals` + the user's recent real prompts), keyed by owner + realpath(workingDir) so they survive `/clear`/respawns and multi-user scoping is structural. Capture rides the transcript (`transcript:user_prompt` from `transcript-watcher.ts`), NOT the input paths: `POST /input` sees only programmatic prompts and the WS channel is raw keystrokes. The listener lives inside `startTranscriptWatcher()`'s `if (!watcher)` block (outside it would duplicate per hook event) and is claude-only + gated on the setting per event. Store: `src/intent-store.ts` singleton, `intents.json` written 0600 tmp+rename (prompts can contain secrets; never fed to `/api/search`). Endpoints: GET/PUT/DELETE `/api/sessions/:id/intent` + POST `/api/sessions/:id/readmymind` (`readmymind-routes.ts`, ownership via `findSessionOrFail` WITH `req`; registrations stay the bare `app.('path')` shape, the endpoints.md drift scanner cannot see generics). **Phase 2 (predictor + 🧠 button)**: `readmymind-context.ts` is the PURE budgeted assembler (9 ranked sources, drop order siblings→away→workspace→tools, sections 1-4 truncate only); IO lives in `readmymind-collectors.ts` (transcript TAIL read — the live watcher keeps only a 500-char snippet — + git signals, skipped for remote-SSH cases) and the route; `readmymind-predictor.ts` reuses the AiCheckerBase spawn mechanics standalone (verdict-shaped base vs freeform JSON) as a mutable singleton routes call and tests stub. Claude-mode only (400), one in flight per session (409 CONFLICT), model = `readMyMindModel` setting defaulting to `AI_CHECK_MODEL` (opus, decided). Frontend `readmymind-ui.js`: header 🧠 marker-hidden (`btn-readmymind--hidden`) until the setting is ON; phones hide it in mobile.css and get a keyboard-accessory 🧠 key instead (ships in BOTH bar templates, revealed by the `rmm-enabled` class on the BAR element — setMode() rebuilds button innerHTML, so per-key state would be wiped; synced at init + every `applyHeaderVisibilitySettings()`). Alternate suggestions render as tappable rows that swap into the editable field without losing edits; Rethink rejects the whole shown set and carries the optional steer note (`#readMyMindSteer`, sent as `steer`, shown in ready + empty-result phases, cleared on each open). Suggestions render via value/`textContent` ONLY and Send/Insert go through `POST /input` (server-side, so the sendEnterKey/local-echo trap does not apply) — nothing auto-sends, ever. User guide: `docs/readmymind.md`.
@@ -298,7 +300,7 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
### Frontend
-Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. Load order: `constants.js`(1) → `i18n.js`(1.5) → `mobile-handlers.js`(2) → `voice-input.js`(3) → `notification-manager.js`(4) → `keyboard-accessory.js`(5) → `input-cjk.js`(5.5) → `terminal-keycode229-recovery.js`(5.55) → `sanitize-html.js`(5.6) → `app.js`(6) → `tab-rail-resize.js`(6.5) → `terminal-ui.js`(7) → `respawn-ui.js`(8) → `ralph-panel.js`(9) → `orchestrator-panel.js`(9.5) → `cron-ui.js`(9.7) → `settings-ui.js`(10) → `panels-ui.js`(11) → `readmymind-ui.js`(11.3) → `ultracode-panel.js`(11.5) → `approvals-ui.js`(11.6) → `admin-ui.js`(11.7) → `session-ui.js`(12) → `webview-tabs.js`(12.5) → `mobile-overview.js`(12.55) → `home-sessions.js`(12.56) → `entrance-animations.js`(12.6) → `ralph-wizard.js`(13) → `api-client.js`(14) → `subagent-windows.js`(15) → `ultracode-windows.js`(15.5) → `session-lineage.js`(15.6) → `image-input.js`(16). `i18n.js` translates static + newly inserted application DOM while skipping terminal/response/file/user-name surfaces; `input-cjk.js` handles CJK IME composition via an always-visible textarea below the terminal (`window.cjkActive` blocks xterm's onData). `terminal-keycode229-recovery.js` forwards a committed `input` event that xterm's `_inputEvent` guard drops (Chrome-on-Android soft keyboards send `composed: true` after a keydown), and only when xterm emitted no canonical data for that keystroke. ⚠️ **That decision is settled at the NEXT keydown as well as on its own zero-delay timer** (#441): the drain runs from xterm's custom key handler, which fires BEFORE xterm processes that key, so a soft keyboard that commits the last character and sends Enter in one InputConnection transaction puts the character on the wire ahead of the `\r`. On the timer alone that character is not merely late, it is LOST: xterm emits the `\r` first and bumps the canonical counter past the candidate's snapshot, so the candidate stands down (measured, `hell\r` where the user typed `hello`). The trade is that a keydown decides with less evidence than the timer did, since xterm's own keyCode-229 rescue has not run yet; that is safe for Enter, which clears the textarea so the pending diff emits nothing. Ordering is pinned by `test/terminal-keycode229-recovery.browser.test.ts`, which the CI gate does NOT run.
+Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. Load order: `constants.js`(1) → `i18n.js`(1.5) → `mobile-handlers.js`(2) → `voice-input.js`(3) → `notification-manager.js`(4) → `keyboard-accessory.js`(5) → `input-cjk.js`(5.5) → `terminal-keycode229-recovery.js`(5.55) → `sanitize-html.js`(5.6) → `app.js`(6) → `tab-rail-resize.js`(6.5) → `terminal-ui.js`(7) → `respawn-ui.js`(8) → `ralph-panel.js`(9) → `orchestrator-panel.js`(9.5) → `cron-ui.js`(9.7) → `settings-ui.js`(10) → `panels-ui.js`(11) → `readmymind-ui.js`(11.3) → `ultracode-panel.js`(11.5) → `approvals-ui.js`(11.6) → `reboot-restore-ui.js`(11.65) → `admin-ui.js`(11.7) → `session-ui.js`(12) → `webview-tabs.js`(12.5) → `mobile-overview.js`(12.55) → `home-sessions.js`(12.56) → `entrance-animations.js`(12.6) → `ralph-wizard.js`(13) → `api-client.js`(14) → `subagent-windows.js`(15) → `ultracode-windows.js`(15.5) → `session-lineage.js`(15.6) → `image-input.js`(16). `i18n.js` translates static + newly inserted application DOM while skipping terminal/response/file/user-name surfaces; `input-cjk.js` handles CJK IME composition via an always-visible textarea below the terminal (`window.cjkActive` blocks xterm's onData). `terminal-keycode229-recovery.js` forwards a committed `input` event that xterm's `_inputEvent` guard drops (Chrome-on-Android soft keyboards send `composed: true` after a keydown), and only when xterm emitted no canonical data for that keystroke. ⚠️ **That decision is settled at the NEXT keydown as well as on its own zero-delay timer** (#441): the drain runs from xterm's custom key handler, which fires BEFORE xterm processes that key, so a soft keyboard that commits the last character and sends Enter in one InputConnection transaction puts the character on the wire ahead of the `\r`. On the timer alone that character is not merely late, it is LOST: xterm emits the `\r` first and bumps the canonical counter past the candidate's snapshot, so the candidate stands down (measured, `hell\r` where the user typed `hello`). The trade is that a keydown decides with less evidence than the timer did, since xterm's own keyCode-229 rescue has not run yet; that is safe for Enter, which clears the textarea so the pending diff emits nothing. Ordering is pinned by `test/terminal-keycode229-recovery.browser.test.ts`, which the CI gate does NOT run.
**Entrance animations** (`entrance-animations.js`, all OFF by default): opt-in animations for the four things that appear when work starts, chosen per surface via `data-tab-anim` / `data-term-anim` / `data-win-anim` / `data-line-anim` on ``. Defaults are the `legacy` theme, so an untouched install behaves exactly as before and every hook short-circuits on its first line. ⚠️ Tabs and connection lines are **destroyed mid-animation** on every re-render (`_fullRenderSessionTabs()` replaces the strip's innerHTML; `_updateConnectionLinesImmediate()` does `svg.innerHTML = ''`), so both are tracked by id and re-applied to the fresh element with a **negative `animation-delay`** to resume rather than restart. ⚠️ The terminal-pane styles may animate **transform / opacity / clip-path only**, xterm's FitAddon derives rows+cols from `getComputedStyle(parent).width/height`, so animating width/height/padding there would resize the PTY; `test/entrance-animations.test.ts` pins that property allowlist, plus the rule→keyframes→theme-option chain a style silently does nothing without. ⚠️ **`blur` is the ONE style that puts a `filter` on the terminal container**, against the standing rule, because every alternative was measured against a live xterm and does not work: a `backdrop-filter` veil on `::before` blurs perfectly while STATIC and Chrome silently drops the backdrop the moment ANY animation runs on that pseudo-element (the veil computes `blur(15.3px)` and the text behind it stays razor sharp), and driving the radius from rAF buys the same full-screen blur per frame plus main-thread work. The cost the rule exists to avoid is inherent to blurring a terminal, so the style buys it knowingly: opt-in, OFF by default, one ~520ms run per session open, class straight back off, `will-change` still unset. Worst-case price, headless SwiftShader with no GPU: frame deltas 16.7ms → 33.3ms for the run, against 16.7ms flat for `fade`. Do not generalise it — a second filtered terminal style needs its own measurement. ⚠️ The `blur` connection line animates `filter` too, so both kinds of line hold their glow in **`--line-glow`** and both of its keyframes say `blur(N) var(--line-glow)`: the function lists then match and interpolate, instead of the glow vanishing for the run and popping back (a lineage line's glow is a different colour entirely, set per element). Its 100% frame deliberately omits `opacity` so the endpoint comes from the element's own resting value — 0.9 subagent, 0.72 lineage, 0.95 working — which is what `line-enter-fade`'s hardcoded 0.9 gets wrong. ⚠️ Window styles other than `beam` transform the window, which moves the rect its connection line is aimed at; `beam` deliberately animates opacity/filter only so its line can draw toward a stable target. Persisted to its own `codeman:*Anim` localStorage keys (per-device, deliberately NOT in the `.strict()` `SettingsUpdateSchema`); picker in App Settings → Appearance, full per-surface lab at `?animlab=1`.
@@ -381,7 +383,7 @@ Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. L
### API Routes
-~232 handlers across 25 route files in `src/web/routes/`: system (56), sessions (34), cases (34), files (17), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), approvals (4), readmymind (4), me (2), teams (2), tab-layout (2), search (1), hooks (1), clipboard (1), status-telemetry (1), voice (1 + the `/ws/voice/stream` relay), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details.
+~233 handlers across 27 route files in `src/web/routes/`: system (56), sessions (34), cases (34), files (17), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), approvals (4), readmymind (4), custom-model (5), reboot-restore (3), me (2), teams (2), tab-layout (2), search (1), hooks (1), clipboard (1), status-telemetry (1), voice (1 + the `/ws/voice/stream` relay), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details.
**HTTP contract** (stable since 0.9.x, see `docs/versioning-policy.md`; full envelope/status/error-code/SSE spec in `docs/api-reference.md`): responses use the `ApiResponse` envelope — `{ success: true, data? }` or `{ success: false, error, errorCode }` (`src/types/api.ts`). `/api/v1/*` is a versioned alias of `/api/*` (URL rewrite in `server.ts`).
diff --git a/docs/api-reference.md b/docs/api-reference.md
index 4bca9768..53b38e8a 100644
--- a/docs/api-reference.md
+++ b/docs/api-reference.md
@@ -66,17 +66,17 @@ The single source of truth is `ErrorStatus` / `httpStatusForErrorCode()` in
`src/types/api.ts`. Clients should branch on `errorCode` (stable) and may rely on
the HTTP status.
-| `errorCode` | HTTP | Meaning |
-| ------------------ | ---- | --------------------------------------------------- |
-| `INVALID_INPUT` | 400 | Malformed request / failed validation |
-| `UNAUTHORIZED` | 401 | Authentication required or failed |
-| `NOT_FOUND` | 404 | Resource does not exist |
-| `SESSION_BUSY` | 409 | Session is busy |
-| `CONFLICT` | 409 | Conflicts with current state (e.g. already running) |
-| `ALREADY_EXISTS` | 409 | Resource already exists |
-| `OPERATION_FAILED` | 422 | Well-formed but could not be completed |
-| `RATE_LIMITED` | 429 | Too many requests |
-| `INTERNAL_ERROR` | 500 | Unexpected server error |
+| `errorCode` | HTTP | Meaning |
+|-------------|------|---------|
+| `INVALID_INPUT` | 400 | Malformed request / failed validation |
+| `UNAUTHORIZED` | 401 | Authentication required or failed |
+| `NOT_FOUND` | 404 | Resource does not exist |
+| `SESSION_BUSY` | 409 | Session is busy |
+| `CONFLICT` | 409 | Conflicts with current state (e.g. already running) |
+| `ALREADY_EXISTS` | 409 | Resource already exists |
+| `OPERATION_FAILED` | 422 | Well-formed but could not be completed |
+| `RATE_LIMITED` | 429 | Too many requests |
+| `INTERNAL_ERROR` | 500 | Unexpected server error |
Adding a new error code is non-breaking; removing or renaming one is a major change.
@@ -87,10 +87,10 @@ exist because SSE is Codeman's only other "tell me when" channel, and an agent
driving the API from a shell tool cannot practically hold a stream and parse
events inline.
-| Call | Blocks until |
-| --------------------------------------------- | -------------------------------------------------- |
-| `GET /api/v1/sessions/:id/wait` | one of a set of lifecycle signals fires |
-| `GET /api/v1/sessions/:id/wait-output` | a literal string appears in the session's output |
+| Call | Blocks until |
+|------|--------------|
+| `GET /api/v1/sessions/:id/wait` | one of a set of lifecycle signals fires |
+| `GET /api/v1/sessions/:id/wait-output` | a literal string appears in the session's output |
| `POST /api/v1/sessions/:id/input` with `wait` | the input is delivered **and then** a signal fires |
`POST .../input` with `wait` is not the same as a `POST` followed by a separate
@@ -140,13 +140,13 @@ contract is a **marker unique to each call** (`MARK="DONE_$RANDOM"`, send
### Signals
-| Signal | Source | Actually fires for |
-| --------- | -------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| `idle` | the session's own `idle` event | `claude`: yes, on ❯-prompt detection after activity. `shell`: **once only**, ~500 ms after start, and never again. External CLIs: not guaranteed (they render their own TUIs and readiness is output stabilization) |
-| `working` | the session's own `working` event | `claude` only in practice (spinner and work-keyword detection are Claude output formats) |
-| `stop` | the Claude Code `stop` hook, the definitive end-of-turn signal | `claude` only |
-| `blocked` | a `permission_prompt` or `elicitation_dialog` hook | `claude` only, and rarer than it looks: see below |
-| `exit` | no process is behind the session | every mode |
+| Signal | Source | Actually fires for |
+|--------|--------|--------------------|
+| `idle` | the session's own `idle` event | `claude`: yes, on ❯-prompt detection after activity. `shell`: **once only**, ~500 ms after start, and never again. External CLIs: not guaranteed (they render their own TUIs and readiness is output stabilization) |
+| `working` | the session's own `working` event | `claude` only in practice (spinner and work-keyword detection are Claude output formats) |
+| `stop` | the Claude Code `stop` hook, the definitive end-of-turn signal | `claude` only |
+| `blocked` | a `permission_prompt` or `elicitation_dialog` hook | `claude` only, and rarer than it looks: see below |
+| `exit` | no process is behind the session | every mode |
`stop` is the signal to orchestrate on where it exists; `idle` is a heuristic
fallback that can flap mid-turn when a spinner pauses. The default set when `until`
@@ -156,12 +156,12 @@ can no longer happen). On a `claude` worker, prefer an explicit `until=stop,exit
once the session is up: the default set's `idle` also resolves on a spinner pause,
and on a fresh session the **startup** `idle` (emitted when the CLI first comes up)
can land inside your first wait window and report a turn that never ran. Measured:
-a session parked on the trust dialog emits no _further_ `idle`, so it is the
+a session parked on the trust dialog emits no *further* `idle`, so it is the
startup transition, not the dialog, that produces the false success below.
⚠️ **`exit` means "nothing is running", which includes "not started yet".** The
server answers from `pid === null` plus a mux-layer pane-death probe, and that
-covers a session that exited — including a worker that died _inside_ its tmux pane
+covers a session that exited — including a worker that died *inside* its tmux pane
while the local attach client (and therefore `pid`) lives on — one that was
detached, and one that was **created but never started**. So the first wait
after `POST /api/v1/sessions` returns `{"signal":"exit","immediate":true}` in
@@ -184,7 +184,7 @@ blocked, and polling `blocked` alone will sit at its timeout.
⚠️ **On a `shell` session, only `exit` and marker-matching are dependable.** A shell
session emits its one `idle` at startup and then stays `status: "idle"` forever,
-whatever the pane is doing, so it never emits a _transition_. Since send-and-wait
+whatever the pane is doing, so it never emits a *transition*. Since send-and-wait
requires a transition (and so does `fresh=1`), both can only time out there:
a documented default `wait` on a shell worker running `sleep 4` times out at the
full 25 s. Synchronize hook-less sessions with `wait-output` and a unique marker
@@ -218,11 +218,11 @@ with `from=buffer` keeps matching long after the dialog is gone. A worked versio
### `GET /api/v1/sessions/:id/wait`
-| Param | Type | Default | Notes |
-| --------- | -------------------------------------------------------- | ---------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| `until` | comma-separated list of `idle,working,stop,blocked,exit` | `stop,idle,exit` | resolves on the first to fire. An unknown token is a `400` naming it, never a silent fallback |
-| `timeout` | positive integer ms | `60000` | **validated first, clamped second.** `0`, a negative value and a fractional value are all `400`s, not clamps; a valid value outside `[1000, 600000]` is clamped and echoed as `wait.timeoutMs` |
-| `fresh` | `0` \| `1` \| `false` \| `true` | `0` | `1` requires an actual transition, ignoring the state at call time |
+| Param | Type | Default | Notes |
+|-------|------|---------|-------|
+| `until` | comma-separated list of `idle,working,stop,blocked,exit` | `stop,idle,exit` | resolves on the first to fire. An unknown token is a `400` naming it, never a silent fallback |
+| `timeout` | positive integer ms | `60000` | **validated first, clamped second.** `0`, a negative value and a fractional value are all `400`s, not clamps; a valid value outside `[1000, 600000]` is clamped and echoed as `wait.timeoutMs` |
+| `fresh` | `0` \| `1` \| `false` \| `true` | `0` | `1` requires an actual transition, ignoring the state at call time |
```bash
curl -s "$API/api/v1/sessions/$SID/wait?until=stop,exit&timeout=60000"
@@ -239,12 +239,12 @@ a plain signal wait, so check the endpoint path before blaming the parameters.
### `GET /api/v1/sessions/:id/wait-output`
-| Param | Type | Default | Notes |
-| --------- | ------------------------------- | -------- | ----------------------------------------------------------------------------------------------------------- |
-| `match` | literal string, 1 to 200 chars | required | substring match against the PTY stream with ANSI escapes stripped. A match spanning two PTY chunks is found |
-| `nocase` | `0` \| `1` \| `false` \| `true` | `0` | case-insensitive compare. The returned snippet keeps the terminal's original casing |
-| `from` | `now` \| `buffer` | `now` | `buffer` scans the tail of the existing terminal buffer (bounded, 256 KB by default) before blocking |
-| `timeout` | positive integer ms | `60000` | same validation and clamp as `/wait` |
+| Param | Type | Default | Notes |
+|-------|------|---------|-------|
+| `match` | literal string, 1 to 200 chars | required | substring match against the PTY stream with ANSI escapes stripped. A match spanning two PTY chunks is found |
+| `nocase` | `0` \| `1` \| `false` \| `true` | `0` | case-insensitive compare. The returned snippet keeps the terminal's original casing |
+| `from` | `now` \| `buffer` | `now` | `buffer` scans the tail of the existing terminal buffer (bounded, 256 KB by default) before blocking |
+| `timeout` | positive integer ms | `60000` | same validation and clamp as `/wait` |
**Matching is literal, never a pattern.** A `regex` parameter is rejected with a
`400` rather than ignored, so a caller that assumed otherwise finds out immediately
@@ -296,10 +296,10 @@ hand-written query string decodes to a space.
Two optional fields on the existing endpoint:
-| Field | Type | Notes |
-| ------------- | ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
-| `wait` | `true` or the same comma grammar as `until` | `true` means the default signal set. Omitted keeps the historical fire-and-forget behavior, unchanged. `null`, `false` and an empty string are all read as **absent**, not as an error and not as "wait for the default" |
-| `waitTimeout` | positive integer ms | same validation **and** clamp as `timeout`: `0`, a negative and a fractional value are `400`s, anything valid is clamped into `[1000, 600000]` and echoed as `wait.timeoutMs` |
+| Field | Type | Notes |
+|-------|------|-------|
+| `wait` | `true` or the same comma grammar as `until` | `true` means the default signal set. Omitted keeps the historical fire-and-forget behavior, unchanged. `null`, `false` and an empty string are all read as **absent**, not as an error and not as "wait for the default" |
+| `waitTimeout` | positive integer ms | same validation **and** clamp as `timeout`: `0`, a negative and a fractional value are `400`s, anything valid is clamped into `[1000, 600000]` and echoed as `wait.timeoutMs` |
Both are `nullish`, so an explicit `null` from `JSON.stringify` is accepted as
"absent" rather than failing validation. That is deliberate: `.optional()` would
@@ -330,24 +330,16 @@ All three nest the wait result under `data.wait`, so one client helper works aga
any of them:
```json
-{
- "success": true,
- "data": {
- "sessionId": "28325fd3-caa7-4178-82bf-87dfebf0f464",
- "status": "idle",
- "limitPaused": false,
- "wait": {
- "signal": "stop",
- "until": ["stop", "idle", "exit"],
- "timedOut": false,
- "immediate": false,
- "ended": false,
- "aborted": false,
- "waitedMs": 8421,
- "timeoutMs": 60000
- }
+{ "success": true, "data": {
+ "sessionId": "28325fd3-caa7-4178-82bf-87dfebf0f464",
+ "status": "idle",
+ "limitPaused": false,
+ "wait": {
+ "signal": "stop", "until": ["stop", "idle", "exit"],
+ "timedOut": false, "immediate": false, "ended": false, "aborted": false,
+ "waitedMs": 8421, "timeoutMs": 60000
}
-}
+}}
```
`POST .../input` returns the same `wait` object alongside `delivered`, `duplicate`,
@@ -361,21 +353,21 @@ redelivery (harmless, the turn it refers to may be long over), while with
client that reads `delivered === false` as "duplicate" silently treats a failed send
as a success.
-| Field | Type | Meaning |
-| ---------------- | ---------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| `wait.signal` | signal \| `null` | the signal that fired (`/wait` and `/input` only) |
-| `wait.until` | array of signals | what the server actually waited on, after narrowing the default set for the session's mode (`/wait` and `/input` only) |
-| `wait.matched` | boolean | the string appeared (`/wait-output` only) |
-| `wait.match` | string | the literal that was searched for (`/wait-output` only) |
-| `wait.snippet` | string \| `null` | bounded window of output around the match, blank runs collapsed for readability (`/wait-output` only) |
-| `wait.timedOut` | boolean | the wait hit its timeout. Still a `200` |
-| `wait.immediate` | boolean | the condition already held at call time, so nothing was waited for (`waitedMs` is 0) |
-| `wait.ended` | boolean | the session went away (deleted or torn down) before the condition was met |
-| `wait.aborted` | boolean | the client hung up, so the waiter was released without resolving — and by that definition a client never reads `true`. When the **server** abandons a wait itself (send-and-wait against a session with no PTY), it answers in about a millisecond with `ended: true`, `delivered: false`, `duplicate: false` and `aborted: false`: `delivered`/`ended` carry that story, and `aborted` stays the transport flag. Present for completeness; treat a `true` as "this wait answered nothing", never as an outcome |
-| `wait.waitedMs` | number | wall-clock ms actually spent waiting |
-| `wait.timeoutMs` | number | the timeout **after clamping**, which is what was applied |
-| `status` | `SessionStatus` | the session's status after the wait, so a caller that timed out still learns where things stand |
-| `limitPaused` | boolean | the session is paused on a usage limit and will emit nothing until its reset, so a timeout here is expected rather than a stall worth retrying hard |
+| Field | Type | Meaning |
+|-------|------|---------|
+| `wait.signal` | signal \| `null` | the signal that fired (`/wait` and `/input` only) |
+| `wait.until` | array of signals | what the server actually waited on, after narrowing the default set for the session's mode (`/wait` and `/input` only) |
+| `wait.matched` | boolean | the string appeared (`/wait-output` only) |
+| `wait.match` | string | the literal that was searched for (`/wait-output` only) |
+| `wait.snippet` | string \| `null` | bounded window of output around the match, blank runs collapsed for readability (`/wait-output` only) |
+| `wait.timedOut` | boolean | the wait hit its timeout. Still a `200` |
+| `wait.immediate` | boolean | the condition already held at call time, so nothing was waited for (`waitedMs` is 0) |
+| `wait.ended` | boolean | the session went away (deleted or torn down) before the condition was met |
+| `wait.aborted` | boolean | the client hung up, so the waiter was released without resolving — and by that definition a client never reads `true`. When the **server** abandons a wait itself (send-and-wait against a session with no PTY), it answers in about a millisecond with `ended: true`, `delivered: false`, `duplicate: false` and `aborted: false`: `delivered`/`ended` carry that story, and `aborted` stays the transport flag. Present for completeness; treat a `true` as "this wait answered nothing", never as an outcome |
+| `wait.waitedMs` | number | wall-clock ms actually spent waiting |
+| `wait.timeoutMs` | number | the timeout **after clamping**, which is what was applied |
+| `status` | `SessionStatus` | the session's status after the wait, so a caller that timed out still learns where things stand |
+| `limitPaused` | boolean | the session is paused on a usage limit and will emit nothing until its reset, so a timeout here is expected rather than a stall worth retrying hard |
Read the outcome by discriminator, in this order:
@@ -398,12 +390,12 @@ read the timeout as "the worker is wedged" and kill a session that was working f
### Errors
-| `errorCode` | HTTP | When |
-| --------------- | ---- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
-| `INVALID_INPUT` | 400 | unknown `until` / `wait` token; `stop` or `blocked` requested explicitly on a mode that installs no hooks (the message names the mode); `regex=` on `/wait-output`; `match` outside 1 to 200 chars; a non-numeric `timeout` |
-| `NOT_FOUND` | 404 | no such session, or one this caller does not own |
-| `SESSION_BUSY` | 409 | this session's waiter cap is full |
-| `RATE_LIMITED` | 429 | a per-owner or process-wide waiter cap is full. Retry later; the session you named is not the problem |
+| `errorCode` | HTTP | When |
+|-------------|------|------|
+| `INVALID_INPUT` | 400 | unknown `until` / `wait` token; `stop` or `blocked` requested explicitly on a mode that installs no hooks (the message names the mode); `regex=` on `/wait-output`; `match` outside 1 to 200 chars; a non-numeric `timeout` |
+| `NOT_FOUND` | 404 | no such session, or one this caller does not own |
+| `SESSION_BUSY` | 409 | this session's waiter cap is full |
+| `RATE_LIMITED` | 429 | a per-owner or process-wide waiter cap is full. Retry later; the session you named is not the problem |
The two capacity codes are deliberately different. A process-wide cap reported as
`SESSION_BUSY` would tell the caller to switch sessions, which cannot help. The
@@ -454,9 +446,9 @@ Design: [`approvals-inbox-plan.md`](approvals-inbox-plan.md).
- `GET /api/v1/approvals` → `{ approvals: ApprovalItem[] }`, oldest first,
ownership-scoped in multi-user mode. `ApprovalItem`: `{ id, sessionId,
-sessionName, kind: 'permission'|'question'|'idle', createdAt, toolName?,
-toolSummary?, message?, cwd?, context?, options?: {n, label}[],
-acknowledgedAt? }`. `context` is the ANSI-stripped visible pane frame;
+ sessionName, kind: 'permission'|'question'|'idle', createdAt, toolName?,
+ toolSummary?, message?, cwd?, context?, options?: {n, label}[],
+ acknowledgedAt? }`. `context` is the ANSI-stripped visible pane frame;
`options` is present only when the dialog's numbered choices parsed
confidently; `acknowledgedAt` marks an item a human has already looked at
(see `/viewed` below) and tells clients not to re-arm its tab alert. Listing
@@ -474,7 +466,7 @@ acknowledgedAt? }`. `context` is the ANSI-stripped visible pane frame;
first, `422 OPERATION_FAILED` when the session refused input.
- `POST /api/v1/approvals/:id/dismiss` removes the item without keystrokes.
- `POST /api/v1/approvals/session/:sessionId/viewed` → `{ sessionId,
-acknowledged: itemId | null }`. Marks the session's pending **idle** item as
+ acknowledged: itemId | null }`. Marks the session's pending **idle** item as
seen by a human (the web UI calls it when you open the session's tab): the
item stays pending and answerable, but stops arming the yellow tab alert on
every client, including after a reload. Permission/question items are never
@@ -498,19 +490,19 @@ heuristic decides whether to ASK, never whether to act.
Claude-mode sessions only (others carry their conversation id in their own
config object); remote and docker sessions are never offered, because both need
another host or container to be up. The plan is in-memory, so a server restart
-drops it and the offer is gone — the conversations themselves are unaffected,
+drops it and the offer is gone; the conversations themselves are unaffected,
since they live in the CLI's own transcript store and stay reachable from the
Resume list. A plan nobody spends expires after 24 hours.
- `GET /api/v1/reboot-restore` → `{ sessions: RestorableSession[],
-scrollbackRestored: false }`, ownership-scoped in multi-user mode.
+ scrollbackRestored: false }`, ownership-scoped in multi-user mode.
`RestorableSession`: `{ id, name?, workingDir, mode, owner? }`. The persisted
record itself is never sent. `scrollbackRestored` is always `false` and exists
so a client states it: a restored session is a NEW pane, so the conversation
continues and the terminal history does not.
- `POST /api/v1/reboot-restore/restore` with `{ sessionIds?: string[] }` (omit
to restore everything the caller can see) → `{ restored: RestorableSession[],
-skipped: { sessionId, reason }[] }`. `reason` is one of `workspace-missing`
+ skipped: { sessionId, reason }[] }`. `reason` is one of `workspace-missing`
(the directory is gone), `workspace-forbidden` (in multi-user mode it is
outside the workspace of the user the session belongs to, re-checked against
that owner's current grant rather than the caller's), `already-live` (the conversation is already
@@ -521,7 +513,7 @@ skipped: { sessionId, reason }[] }`. `reason` is one of `workspace-missing`
removed from the plan before any pane is built, so a double-click cannot put
two panes on one conversation; anything that never became a pane goes back on
offer, except `already-live`, which cannot stop being true. A restored session
- comes back attached, idle and disarmed — respawn controllers and Ralph loops
+ comes back attached, idle and disarmed: respawn controllers and Ralph loops
are never re-armed automatically.
- `POST /api/v1/reboot-restore/dismiss` → `{ dismissed: n }`. Drops the offer
for everything the caller can see.
@@ -541,7 +533,7 @@ user guide: [`readmymind.md`](readmymind.md).
- `GET /api/v1/sessions/:id/intent` -> `{ intent: IntentProfile }` for the
session's case. `IntentProfile`: `{ key, workingDir, updatedAt, goals,
-recentPrompts: { ts, sessionId, text }[] }` (prompts oldest first, FIFO cap
+ recentPrompts: { ts, sessionId, text }[] }` (prompts oldest first, FIFO cap
50, each <= 500 chars). A case with nothing recorded answers an empty
profile with `updatedAt: 0`; nothing is persisted by reads.
- `PUT /api/v1/sessions/:id/intent` with `{ goals }` (<= 8192 chars, strict
@@ -574,7 +566,7 @@ same speech-to-text service the CLI's own `/voice` mode uses. Gated on the synce
[`claude-voice-plan.md`](claude-voice-plan.md).
- `GET /api/v1/voice/status` -> `{ available, reason?, subscriptionType?,
-expiresAt? }`. `reason` is `disabled` (setting off), `no-credentials` (nobody
+ expiresAt? }`. `reason` is `disabled` (setting off), `no-credentials` (nobody
signed in to Claude Code on the server), `expired` (the access token elapsed;
running any Claude session refreshes it) or `malformed`. The OAuth token
itself is never returned by this or any other endpoint.
diff --git a/src/reboot-restore.ts b/src/reboot-restore.ts
index 85760e8d..40b29b08 100644
--- a/src/reboot-restore.ts
+++ b/src/reboot-restore.ts
@@ -72,11 +72,17 @@ export interface RebootEvidence {
* This heuristic decides whether to ASK, never whether to act. A wrong yes costs
* the user a banner they dismiss, because the restore itself waits for a click.
*
- * ⚠️ `os.uptime()` reports the HOST's uptime, which a container shares. A Codeman
- * running in Docker therefore sees a long uptime after its own container restarts,
- * the boot test fails, and no banner appears. The feature is effectively off for
- * containerized installs. That is the safe direction to fail in, and fixing it
- * needs a boot signal the container actually owns rather than a wider heuristic.
+ * ⚠️ `os.uptime()` reports the HOST's uptime, which a container shares, and that
+ * cuts BOTH ways rather than simply switching the feature off in Docker. After a
+ * genuine host reboot a containerized Codeman sees the host's short uptime, so the
+ * banner DOES appear and the feature works. What it cannot see is a container-only
+ * restart: the host uptime is long, the boot test fails, and no banner appears
+ * although every in-container pane is gone (`docker/server.Dockerfile` installs
+ * tmux inside the Codeman container, and the self-updater restarts the Compose
+ * deployment by exiting the container, so that is the case where this would help
+ * most). Failing quiet is the safe direction, and closing the gap needs a boot
+ * signal the container owns (PID 1's start time, gated on the existing
+ * `isRunningInContainer()`) rather than a wider heuristic.
*/
export function looksLikeHostReboot(evidence: RebootEvidence): boolean {
if (evidence.deadSessionCount === 0) return false;
diff --git a/src/web/ports/session-port.ts b/src/web/ports/session-port.ts
index 85add1d8..78e1fbbd 100644
--- a/src/web/ports/session-port.ts
+++ b/src/web/ports/session-port.ts
@@ -27,7 +27,19 @@ export interface SessionPort {
reapplyPersistedSessionState(
session: Session,
saved: SessionState,
- phase: 'before-spawn' | 'after-spawn'
+ phase: 'before-spawn' | 'after-spawn',
+ options?: {
+ /**
+ * Re-arm a PENDING auto-resume schedule from the record's `autoResumeAt`.
+ * Default true, which is what a Codeman restart wants: the limit footer
+ * will not reprint on its own, so dropping the stamp there strands the
+ * pause. A reboot restore passes false: the stamp predates the reboot,
+ * the pane is new, and re-arming means every restored session types
+ * `continue` into itself about a minute after one click. Auto-resume
+ * stays ENABLED either way, so it re-arms on fresh evidence.
+ */
+ rearmAutoResumeSchedule?: boolean;
+ }
): Promise;
/**
* Undo a session that was registered but never got a working pane: the map
diff --git a/src/web/public/reboot-restore-ui.js b/src/web/public/reboot-restore-ui.js
index 2a3599dd..1e738bdd 100644
--- a/src/web/public/reboot-restore-ui.js
+++ b/src/web/public/reboot-restore-ui.js
@@ -26,7 +26,7 @@
* @mixin Extends CodemanApp.prototype via Object.assign
* @dependency app.js (CodemanApp class, showToast)
* @dependency api-client.js at runtime (this._api / this._apiJson)
- * @loadorder 11.7 of 17, after approvals-ui.js
+ * @loadorder 11.65, after approvals-ui.js and before admin-ui.js (11.7)
*/
/** Plain-language wording for one skip reason, for the toast after a restore. */
diff --git a/src/web/public/styles.css b/src/web/public/styles.css
index f35bf5cd..975b4d8b 100644
--- a/src/web/public/styles.css
+++ b/src/web/public/styles.css
@@ -2548,6 +2548,10 @@ body.solo-mode .header-tokens,
body.solo-mode .btn-notifications,
body.solo-mode .btn-multimonitor,
body.solo-mode .header-plan-usage,
+/* A solo window shows ONE session and has no tab strip to put restored ones in,
+ so offering to rebuild a list of them there is an offer it cannot show the
+ result of. The dashboard that spawned this window carries the banner. */
+body.solo-mode .reboot-restore-banner,
body.solo-mode .btn-lifecycle-log {
display: none !important;
}
diff --git a/src/web/routes/reboot-restore-routes.ts b/src/web/routes/reboot-restore-routes.ts
index 3ff27b4f..fb6cec97 100644
--- a/src/web/routes/reboot-restore-routes.ts
+++ b/src/web/routes/reboot-restore-routes.ts
@@ -41,7 +41,7 @@ import { clampEnvOverridesForOwner } from '../../session-env-clamp.js';
import { Session } from '../../session.js';
import { resolveClaudeModeForUsername } from '../../user-store.js';
import { getCli } from '../../config/cli-registry/registry.js';
-import { applyWorkspaceHooks } from '../../hooks-config.js';
+import { applyWorkspaceHooks, seedAgentSessionPreamble } from '../../hooks-config.js';
import { getLifecycleLog } from '../../session-lifecycle-log.js';
import { STATS_COLLECTION_INTERVAL_MS } from '../../config/server-timing.js';
import { SseEvent } from '../sse-events.js';
@@ -105,11 +105,16 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
// the user resumed by hand from the Resume list is already on screen, and a
// second pane on it would fight the first for the same transcript. This one
// is never re-offered: unlike a missing workspace, it cannot stop being true.
- const liveSessionIds = new Set(ctx.sessions.keys());
- const liveConversationIds = new Set(
- [...ctx.sessions.values()].map((session) => session.claudeSessionId).filter((id): id is string => !!id)
- );
- const { restore, skipped } = rejectAlreadyLive(taken, liveSessionIds, liveConversationIds);
+ // Read fresh each time rather than snapshotted once: the loop below awaits a
+ // real `startInteractive()` per entry, so by the tenth entry a snapshot taken
+ // here is tens of seconds old, and a conversation the user resumed by hand in
+ // that window would be invisible to it.
+ const liveSessionIds = () => new Set(ctx.sessions.keys());
+ const liveConversationIds = () =>
+ new Set(
+ [...ctx.sessions.values()].map((session) => session.claudeSessionId).filter((id): id is string => !!id)
+ );
+ const { restore, skipped } = rejectAlreadyLive(taken, liveSessionIds(), liveConversationIds());
for (const entry of taken) {
if (skipped.some((s) => s.sessionId === entry.sessionId)) unspent.delete(entry);
}
@@ -119,6 +124,18 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
const workspaceHooksEnabled = await ctx.getWorkspaceHooksEnabled();
for (const entry of restore) {
+ // The already-live check, re-run against the board as it is NOW. The pass
+ // above decided the batch; this catches a conversation that went live while
+ // an earlier entry in this same batch was starting. Spent rather than
+ // returned to the plan, for the same reason as the batch pass: unlike a
+ // missing workspace or a withdrawn grant, an open conversation is not a
+ // condition that stops being true.
+ const [lateLive] = rejectAlreadyLive([entry], liveSessionIds(), liveConversationIds()).skipped;
+ if (lateLive) {
+ failures.push(lateLive);
+ unspent.delete(entry);
+ continue;
+ }
// Capacity is re-checked per iteration, because this loop is itself
// creating the sessions it counts. The offer can be a day old, so the
// board may be fuller now than the plan assumed.
@@ -157,6 +174,12 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
workingDir: saved.workingDir,
mode: saved.mode,
name: saved.name,
+ // Without this the constructor re-infers ownership from the name, so a
+ // session the user renamed by hand to something shaped like `w-`
+ // comes back as `placeholder` and auto-naming overwrites their name on
+ // the next prompt. The route persists below, so the loss would go to
+ // disk. `restoreMuxSessions()` passes it for the same reason.
+ nameSource: saved.nameSource,
createdAt: saved.createdAt,
mux: ctx.mux,
useMux: true,
@@ -197,7 +220,14 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
// the reduced one and drop the pin that keeps it from being pruned. A
// listener-driven persist can still land inside the debounce window
// while the pane starts; the write below repairs the record.
- await ctx.reapplyPersistedSessionState(session, saved, 'after-spawn');
+ // `rearmAutoResumeSchedule: false`: the saved stamp predates the reboot and
+ // the pane is new, so honouring it would have every restored session type
+ // `continue` into itself about a minute after one click. Auto-resume stays
+ // enabled and re-arms on the next real limit message. This is also what the
+ // module header promises ("comes back attached, idle and disarmed").
+ await ctx.reapplyPersistedSessionState(session, saved, 'after-spawn', {
+ rearmAutoResumeSchedule: false,
+ });
ctx.persistSessionState(session);
// A session without its workspace hooks goes silently blind: no stop or
@@ -211,6 +241,16 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes
);
}
+ // Both create paths seed this; without it a restored claude session's agent
+ // skill falls back to writing out the whole ~150-line §0 preamble. Remote and
+ // docker sessions never reach here (the plan rejects them as
+ // `remote-or-docker`), so the local-only condition is structural.
+ if (getCli(session.mode)?.capabilities.agentSkillInjection && (await ctx.getAgentSkillEnabled())) {
+ await seedAgentSessionPreamble(session.id).catch((err: unknown) =>
+ console.warn(`[agent-skill] preamble seed failed for ${session.id}: ${getErrorMessage(err)}`)
+ );
+ }
+
getLifecycleLog().log({ event: 'recovered', sessionId: session.id, name: session.name });
// Every other open tab and phone needs this; the clicking tab already has
// the response, and the client's handler is an idempotent upsert.
diff --git a/src/web/server.ts b/src/web/server.ts
index e94b8cdf..3497afdf 100644
--- a/src/web/server.ts
+++ b/src/web/server.ts
@@ -2940,7 +2940,8 @@ export class WebServer extends EventEmitter {
async reapplyPersistedSessionState(
session: Session,
saved: SessionState,
- phase: 'before-spawn' | 'after-spawn'
+ phase: 'before-spawn' | 'after-spawn',
+ options?: { rearmAutoResumeSchedule?: boolean }
): Promise {
if (phase === 'before-spawn') {
// The custom-model env has to be rebuilt from the endpoint store: the persist
@@ -2968,7 +2969,14 @@ export class WebServer extends EventEmitter {
session.setAutoClear(saved.autoClearEnabled ?? false, saved.autoClearThreshold);
}
if (saved.autoResumeEnabled) {
- session.restoreAutoResume(true, saved.autoResumeAt);
+ // The stamp is re-armed by default, because a Codeman restart leaves the
+ // limit footer un-reprinted and dropping it there would strand the pause.
+ // A reboot restore opts out: that stamp predates the reboot, the pane is
+ // new, and honouring it means every session the user restored types
+ // `continue` into itself about a minute later, unattended. The setting
+ // itself stays on either way, so it re-arms on the next limit message.
+ const rearm = options?.rearmAutoResumeSchedule !== false;
+ session.restoreAutoResume(true, rearm ? saved.autoResumeAt : undefined);
}
if (saved.inputTokens !== undefined || saved.outputTokens !== undefined || saved.totalCost !== undefined) {
session.restoreTokens(saved.inputTokens ?? 0, saved.outputTokens ?? 0, saved.totalCost ?? 0);
@@ -3024,9 +3032,18 @@ export class WebServer extends EventEmitter {
session.ralphTracker.stopWatchingFixPlan();
const summaryTracker = this.runSummaryTrackers.get(sessionId);
if (summaryTracker) {
+ // Closes the run's own record before the tracker goes, the way
+ // `_doCleanupSession()` does. Cosmetic rather than load-bearing, but a
+ // run left open reads as still going in the away digest.
+ summaryTracker.recordSessionStopped();
summaryTracker.stop();
this.runSummaryTrackers.delete(sessionId);
}
+ // Also mirrors `_doCleanupSession()`. The PERSISTED Ralph state is left
+ // alone on purpose (that is one of the things separating this from
+ // cleanupSession); this only clears the in-memory tracker the failed
+ // construction built, which the retry reuses the id of.
+ session.ralphTracker.fullReset();
// --- what anything else may have attached to this id in the meantime ---
// A rebuild can fail AFTER startInteractive() resolved, and a restored
diff --git a/test/routes/reboot-restore-routes.test.ts b/test/routes/reboot-restore-routes.test.ts
index dcb7d7ed..19067d21 100644
--- a/test/routes/reboot-restore-routes.test.ts
+++ b/test/routes/reboot-restore-routes.test.ts
@@ -11,7 +11,10 @@
* The routes read the process-wide `rebootRestoreRegistry` singleton, so every
* test resets it; a leaked entry would bleed into the next one.
*/
-import { describe, it, expect, afterEach } from 'vitest';
+import { describe, it, expect, afterEach, beforeEach } from 'vitest';
+import { mkdtempSync, rmSync } from 'node:fs';
+import { tmpdir } from 'node:os';
+import { join } from 'node:path';
import Fastify, { type FastifyInstance } from 'fastify';
import fastifyCookie from '@fastify/cookie';
import { registerRebootRestoreRoutes } from '../../src/web/routes/reboot-restore-routes.js';
@@ -187,6 +190,49 @@ describe('POST /api/reboot-restore/restore', () => {
});
});
+describe('POST /api/reboot-restore/restore: multi-user workspace confinement', () => {
+ const saved: Record = {};
+ let realDir: string;
+
+ beforeEach(() => {
+ saved.CODEMAN_MULTIUSER = process.env.CODEMAN_MULTIUSER;
+ process.env.CODEMAN_MULTIUSER = '1';
+ // This branch sits AFTER the existsSync check, so the workspace has to be
+ // real for the confinement rule to be the thing that rejects the entry.
+ realDir = mkdtempSync(join(tmpdir(), 'codeman-reboot-restore-real-'));
+ });
+
+ afterEach(() => {
+ if (saved.CODEMAN_MULTIUSER === undefined) delete process.env.CODEMAN_MULTIUSER;
+ else process.env.CODEMAN_MULTIUSER = saved.CODEMAN_MULTIUSER;
+ rmSync(realDir, { recursive: true, force: true });
+ });
+
+ it("refuses a workspace outside the OWNER's case space, and leaves it on offer", async () => {
+ const entry = offerEntry('a', 'alice');
+ entry.workingDir = realDir;
+ (entry.state as { workingDir: string }).workingDir = realDir;
+ rebootRestoreRegistry.set([entry]);
+
+ // An admin does the clicking. The confinement is still resolved against
+ // alice, the entry's OWNER: `isWorkingDirAllowed` waves an admin through, so
+ // reading the caller here would hand an admin the power to rebuild another
+ // user's session anywhere on the box.
+ const app = await createHarness({ username: 'root-user', role: 'admin' });
+ const res = await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} });
+
+ expect(res.statusCode).toBe(200);
+ expect(res.json().data.restored).toEqual([]);
+ expect(res.json().data.skipped).toEqual([{ sessionId: 'a', reason: 'workspace-forbidden' }]);
+
+ // A withdrawn grant can be given back, so unlike `already-live` this is not
+ // the permanent kind of refusal and the entry stays claimable.
+ const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data;
+ expect(left.sessions.map((s: { id: string }) => s.id)).toEqual(['a']);
+ await app.close();
+ });
+});
+
describe('POST /api/reboot-restore/dismiss', () => {
it('drops the offer and leaves the banner with nothing to show', async () => {
rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]);
From 0e1191b774ad7f1ac32696ee0a0d9de1d3496904 Mon Sep 17 00:00:00 2001
From: Codeman maintainer
Date: Fri, 18 Sep 2026 13:55:22 +0200
Subject: [PATCH 23/28] chore(changeset): trim the contributor entries and add
the 1.30.0 thanks
Changeset text becomes user-facing CHANGELOG, so the #429 entry is cut
from five bullets of internal bash-array detail down to what the change
does for someone running the installer, as promised on the PR. The #441
entry loses its em-dashes, which are not house style. Adds an entry for
the maintainer fixes applied while landing #442, and the Thanks section.
Co-Authored-By: Claude Opus 5 (1M context)
---
.changeset/android-last-character-on-enter.md | 2 +-
.changeset/cli-catalog-followups.md | 21 +------------------
.changeset/release-1-30-0-merge-fixes.md | 5 +++++
.changeset/release-1-30-0-thanks.md | 9 ++++++++
4 files changed, 16 insertions(+), 21 deletions(-)
create mode 100644 .changeset/release-1-30-0-merge-fixes.md
create mode 100644 .changeset/release-1-30-0-thanks.md
diff --git a/.changeset/android-last-character-on-enter.md b/.changeset/android-last-character-on-enter.md
index 0af4c14a..c68df755 100644
--- a/.changeset/android-last-character-on-enter.md
+++ b/.changeset/android-last-character-on-enter.md
@@ -2,4 +2,4 @@
"aicodeman": patch
---
-Stop a phone keyboard losing the last character of every message it sends. Android soft keyboards commit the last typed character and send the Enter key in one InputConnection transaction, so the `input` event and the Enter keydown are both processed before any zero-delay timer runs. The orphaned-input recovery from #388 only resolved its candidate on such a timer, and lost it both ways: xterm emits `\r` synchronously from the Enter keydown, so the local-echo composer submitted the prompt before the recovered character existed, and that `\r` bumped the "did xterm speak for this keystroke" counter, so the candidate then stood itself down and dropped the character outright. Pending candidates are now drained synchronously at the next keydown, from xterm's custom key handler — before xterm processes that key — so the counter still holds the value it had while the candidate's own keystroke was current, and the recovered byte reaches the composer ahead of the Enter. Typing on a physical keyboard is unaffected: there, the timer has already resolved the candidate before the next key arrives.
+Stop a phone keyboard losing the last character of every message it sends. Android soft keyboards commit the last typed character and send the Enter key in one InputConnection transaction, so the `input` event and the Enter keydown are both processed before any zero-delay timer runs. The orphaned-input recovery from #388 only resolved its candidate on such a timer, and lost it both ways: xterm emits `\r` synchronously from the Enter keydown, so the local-echo composer submitted the prompt before the recovered character existed, and that `\r` bumped the "did xterm speak for this keystroke" counter, so the candidate then stood itself down and dropped the character outright. Pending candidates are now drained synchronously at the next keydown, from xterm's custom key handler, which runs before xterm processes that key, so the counter still holds the value it had while the candidate's own keystroke was current, and the recovered byte reaches the composer ahead of the Enter. Typing on a physical keyboard is unaffected: there, the timer has already resolved the candidate before the next key arrives.
diff --git a/.changeset/cli-catalog-followups.md b/.changeset/cli-catalog-followups.md
index 5970e119..1e16cbcf 100644
--- a/.changeset/cli-catalog-followups.md
+++ b/.changeset/cli-catalog-followups.md
@@ -2,23 +2,4 @@
"aicodeman": patch
---
-Cleans up the loose ends the maintainer flagged as "worth knowing rather than fixing" when
-merging the CLI-catalogue-driven `install.sh`/Docker-agent-image PR (#380):
-
-- `install.sh` no longer carries `_cli_index`/`check_cli`/`get_cli_path`, three generic
- lookup helpers left behind once the catalogue-driven menu and hints stopped calling them.
-- The generator no longer emits `CLI_KIND`/`CLI_NPM`, two bash arrays nothing in `install.sh`
- read (the `.mjs`/`docker-hosts.ts` producers already read the JSON catalogue's `kind`/
- `npmPackage` fields directly).
-- `detect_all_clis` now skips a disabled entry entirely rather than probing it and filtering
- the result downstream — no stock entry ships disabled today, so this is a latent
- inefficiency closed before it is a latent bug, not a behaviour change.
-- The install hint for a `launcherProfile` entry (DeepSeek today) now explains, in one line,
- why it is a docs link rather than a runnable command — its docs page documents
- `npm install -g @deepseek-ai/dsh`, which installs the launcher only and cannot drive a pane
- on its own, the exact trap the menu already avoids by withholding the command. Driven by a
- new generated `CLI_LAUNCHER_ONLY` array (from `discovery.launcherProfile`), not an id check.
-- The non-interactive default's comment no longer claims it is always Claude Code: on a
- wget-only host, Claude's curl one-liner is filtered out of the offered list first, so the
- default becomes whichever npm-based entry sorts earliest instead. Behaviour is unchanged
- (and was already printed, so never silent) — only the comment was wrong.
+The installer's hint for a launcher-only CLI (DeepSeek today) now says why it is a docs link rather than a command you can run, and points at the thing that resolves it: the package installs a launcher that still needs a terminal profile, and Codeman's Run menu can add one in a click. Driven by a generated `CLI_LAUNCHER_ONLY` flag rather than an id check, so it covers any future entry of that shape. Also removes three dead lookup helpers and two never-read generated arrays from `install.sh`, skips a disabled entry's probe instead of filtering it afterwards, and corrects a comment that claimed the non-interactive default is always Claude Code (on a wget-only host its curl one-liner is filtered out first).
diff --git a/.changeset/release-1-30-0-merge-fixes.md b/.changeset/release-1-30-0-merge-fixes.md
new file mode 100644
index 00000000..71f6e65b
--- /dev/null
+++ b/.changeset/release-1-30-0-merge-fixes.md
@@ -0,0 +1,5 @@
+---
+"aicodeman": patch
+---
+
+Maintainer fixes applied while landing the above. A session restored after a reboot keeps the name you gave it (the rebuild dropped the field that records who named a session, so a hand-renamed session came back looking auto-named and the next prompt overwrote it), and no longer types `continue` into itself on its own: a pending auto-resume stamp from before the reboot is dropped rather than re-armed, since the pane is new and one click could otherwise arm several unattended prompts at once. Auto-resume itself stays on and re-arms on the next real usage-limit message. The restore offer is also hidden in a detached single-session window, which has no tab strip to put restored sessions in, and a conversation that goes live while an earlier session in the same batch is starting is no longer restored a second time.
diff --git a/.changeset/release-1-30-0-thanks.md b/.changeset/release-1-30-0-thanks.md
new file mode 100644
index 00000000..767cad6c
--- /dev/null
+++ b/.changeset/release-1-30-0-thanks.md
@@ -0,0 +1,9 @@
+---
+"aicodeman": patch
+---
+
+### Thanks
+
+- @irisitymichaelgrundberg for the reboot-restore banner (#442), and for the three real reboots behind it rather than a mocked one.
+- @shenlvkang-collab for tracking down why Android keyboards lost the last character of every message (#441), including the half where the character was not late but gone.
+- @opticon454 for going back and closing out the loose ends left as "worth knowing rather than fixing" after #380 (#429).
From 20fc7b3c3d25bc5646f1dda0f02b20c1d40e1bd0 Mon Sep 17 00:00:00 2001
From: "github-actions[bot]"
<41898282+github-actions[bot]@users.noreply.github.com>
Date: Fri, 18 Sep 2026 14:09:55 +0200
Subject: [PATCH 24/28] chore: version packages (#447)
* chore: version packages
* chore: sync the CLAUDE.md version line to 1.30.0
The changesets bot does not touch this line, and pushing it to master
after merging the version PR starts a second Release run that has raced
the first before. Riding the bot's own branch keeps it to one push.
Co-Authored-By: Claude Opus 5 (1M context)
---------
Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>
Co-authored-by: Codeman maintainer
---
.changeset/android-last-character-on-enter.md | 5 -----
.changeset/cli-catalog-followups.md | 5 -----
.changeset/reboot-restore-banner.md | 5 -----
.changeset/release-1-30-0-merge-fixes.md | 5 -----
.changeset/release-1-30-0-thanks.md | 9 ---------
.../terminal-history-anchor-after-parse.md | 5 -----
.claude-plugin/marketplace.json | 2 +-
CHANGELOG.md | 18 ++++++++++++++++++
CLAUDE.md | 2 +-
package-lock.json | 4 ++--
package.json | 2 +-
plugins/codeman/.claude-plugin/plugin.json | 2 +-
12 files changed, 24 insertions(+), 40 deletions(-)
delete mode 100644 .changeset/android-last-character-on-enter.md
delete mode 100644 .changeset/cli-catalog-followups.md
delete mode 100644 .changeset/reboot-restore-banner.md
delete mode 100644 .changeset/release-1-30-0-merge-fixes.md
delete mode 100644 .changeset/release-1-30-0-thanks.md
delete mode 100644 .changeset/terminal-history-anchor-after-parse.md
diff --git a/.changeset/android-last-character-on-enter.md b/.changeset/android-last-character-on-enter.md
deleted file mode 100644
index c68df755..00000000
--- a/.changeset/android-last-character-on-enter.md
+++ /dev/null
@@ -1,5 +0,0 @@
----
-"aicodeman": patch
----
-
-Stop a phone keyboard losing the last character of every message it sends. Android soft keyboards commit the last typed character and send the Enter key in one InputConnection transaction, so the `input` event and the Enter keydown are both processed before any zero-delay timer runs. The orphaned-input recovery from #388 only resolved its candidate on such a timer, and lost it both ways: xterm emits `\r` synchronously from the Enter keydown, so the local-echo composer submitted the prompt before the recovered character existed, and that `\r` bumped the "did xterm speak for this keystroke" counter, so the candidate then stood itself down and dropped the character outright. Pending candidates are now drained synchronously at the next keydown, from xterm's custom key handler, which runs before xterm processes that key, so the counter still holds the value it had while the candidate's own keystroke was current, and the recovered byte reaches the composer ahead of the Enter. Typing on a physical keyboard is unaffected: there, the timer has already resolved the candidate before the next key arrives.
diff --git a/.changeset/cli-catalog-followups.md b/.changeset/cli-catalog-followups.md
deleted file mode 100644
index 1e16cbcf..00000000
--- a/.changeset/cli-catalog-followups.md
+++ /dev/null
@@ -1,5 +0,0 @@
----
-"aicodeman": patch
----
-
-The installer's hint for a launcher-only CLI (DeepSeek today) now says why it is a docs link rather than a command you can run, and points at the thing that resolves it: the package installs a launcher that still needs a terminal profile, and Codeman's Run menu can add one in a click. Driven by a generated `CLI_LAUNCHER_ONLY` flag rather than an id check, so it covers any future entry of that shape. Also removes three dead lookup helpers and two never-read generated arrays from `install.sh`, skips a disabled entry's probe instead of filtering it afterwards, and corrects a comment that claimed the non-interactive default is always Claude Code (on a wget-only host its curl one-liner is filtered out first).
diff --git a/.changeset/reboot-restore-banner.md b/.changeset/reboot-restore-banner.md
deleted file mode 100644
index 33942048..00000000
--- a/.changeset/reboot-restore-banner.md
+++ /dev/null
@@ -1,5 +0,0 @@
----
-'aicodeman': minor
----
-
-Offer to rebuild the sessions a host reboot destroyed. A reboot takes the tmux server down with it, so every pane dies and the board comes up empty. Codeman now works out what was running, and the board offers to restore it behind a click. The conversations come back; the terminal scrollback does not, and the banner says so.
diff --git a/.changeset/release-1-30-0-merge-fixes.md b/.changeset/release-1-30-0-merge-fixes.md
deleted file mode 100644
index 71f6e65b..00000000
--- a/.changeset/release-1-30-0-merge-fixes.md
+++ /dev/null
@@ -1,5 +0,0 @@
----
-"aicodeman": patch
----
-
-Maintainer fixes applied while landing the above. A session restored after a reboot keeps the name you gave it (the rebuild dropped the field that records who named a session, so a hand-renamed session came back looking auto-named and the next prompt overwrote it), and no longer types `continue` into itself on its own: a pending auto-resume stamp from before the reboot is dropped rather than re-armed, since the pane is new and one click could otherwise arm several unattended prompts at once. Auto-resume itself stays on and re-arms on the next real usage-limit message. The restore offer is also hidden in a detached single-session window, which has no tab strip to put restored sessions in, and a conversation that goes live while an earlier session in the same batch is starting is no longer restored a second time.
diff --git a/.changeset/release-1-30-0-thanks.md b/.changeset/release-1-30-0-thanks.md
deleted file mode 100644
index 767cad6c..00000000
--- a/.changeset/release-1-30-0-thanks.md
+++ /dev/null
@@ -1,9 +0,0 @@
----
-"aicodeman": patch
----
-
-### Thanks
-
-- @irisitymichaelgrundberg for the reboot-restore banner (#442), and for the three real reboots behind it rather than a mocked one.
-- @shenlvkang-collab for tracking down why Android keyboards lost the last character of every message (#441), including the half where the character was not late but gone.
-- @opticon454 for going back and closing out the loose ends left as "worth knowing rather than fixing" after #380 (#429).
diff --git a/.changeset/terminal-history-anchor-after-parse.md b/.changeset/terminal-history-anchor-after-parse.md
deleted file mode 100644
index 418171c4..00000000
--- a/.changeset/terminal-history-anchor-after-parse.md
+++ /dev/null
@@ -1,5 +0,0 @@
----
-"aicodeman": patch
----
-
-Keep the terminal anchored where you are reading while an agent streams (#358). Scrolling up during a Codex response could still be dragged back to the live bottom by the next redraw: the flush captured the viewport before writing and restored it immediately after, but xterm parses asynchronously, so at that moment the buffer had not moved yet, the restore compared the anchor against itself and did nothing, and the redraw landed a tick later with nothing left to pull the view back. The restore now runs inside xterm's own write callback, which is the first point at which the redraw's effect exists, and it holds across consecutive and chunked redraws. It is dropped if you switch sessions or a history replay starts before the write parses, since the anchor indexes the buffer it was captured from.
diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json
index 1d5433b4..7402d884 100644
--- a/.claude-plugin/marketplace.json
+++ b/.claude-plugin/marketplace.json
@@ -10,7 +10,7 @@
"name": "codeman",
"source": "./plugins/codeman",
"description": "Drive Codeman from inside a Claude Code session: spawn worker sessions, prompt them, wait for them, read their answers, clean up. Acts only inside a Codeman-managed session.",
- "version": "1.29.1",
+ "version": "1.30.0",
"author": {
"name": "Ark0N",
"url": "https://github.com/Ark0N"
diff --git a/CHANGELOG.md b/CHANGELOG.md
index c2740d9e..10fdcd21 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -1,5 +1,23 @@
# aicodeman
+## 1.30.0
+
+### Minor Changes
+
+- da933d7: Offer to rebuild the sessions a host reboot destroyed. A reboot takes the tmux server down with it, so every pane dies and the board comes up empty. Codeman now works out what was running, and the board offers to restore it behind a click. The conversations come back; the terminal scrollback does not, and the banner says so.
+
+### Patch Changes
+
+- a1c35da: Stop a phone keyboard losing the last character of every message it sends. Android soft keyboards commit the last typed character and send the Enter key in one InputConnection transaction, so the `input` event and the Enter keydown are both processed before any zero-delay timer runs. The orphaned-input recovery from #388 only resolved its candidate on such a timer, and lost it both ways: xterm emits `\r` synchronously from the Enter keydown, so the local-echo composer submitted the prompt before the recovered character existed, and that `\r` bumped the "did xterm speak for this keystroke" counter, so the candidate then stood itself down and dropped the character outright. Pending candidates are now drained synchronously at the next keydown, from xterm's custom key handler, which runs before xterm processes that key, so the counter still holds the value it had while the candidate's own keystroke was current, and the recovered byte reaches the composer ahead of the Enter. Typing on a physical keyboard is unaffected: there, the timer has already resolved the candidate before the next key arrives.
+- 3f2928a: The installer's hint for a launcher-only CLI (DeepSeek today) now says why it is a docs link rather than a command you can run, and points at the thing that resolves it: the package installs a launcher that still needs a terminal profile, and Codeman's Run menu can add one in a click. Driven by a generated `CLI_LAUNCHER_ONLY` flag rather than an id check, so it covers any future entry of that shape. Also removes three dead lookup helpers and two never-read generated arrays from `install.sh`, skips a disabled entry's probe instead of filtering it afterwards, and corrects a comment that claimed the non-interactive default is always Claude Code (on a wget-only host its curl one-liner is filtered out first).
+- 0e1191b: Maintainer fixes applied while landing the above. A session restored after a reboot keeps the name you gave it (the rebuild dropped the field that records who named a session, so a hand-renamed session came back looking auto-named and the next prompt overwrote it), and no longer types `continue` into itself on its own: a pending auto-resume stamp from before the reboot is dropped rather than re-armed, since the pane is new and one click could otherwise arm several unattended prompts at once. Auto-resume itself stays on and re-arms on the next real usage-limit message. The restore offer is also hidden in a detached single-session window, which has no tab strip to put restored sessions in, and a conversation that goes live while an earlier session in the same batch is starting is no longer restored a second time.
+- 0e1191b: ### Thanks
+ - @irisitymichaelgrundberg for the reboot-restore banner (#442), and for the three real reboots behind it rather than a mocked one.
+ - @shenlvkang-collab for tracking down why Android keyboards lost the last character of every message (#441), including the half where the character was not late but gone.
+ - @opticon454 for going back and closing out the loose ends left as "worth knowing rather than fixing" after #380 (#429).
+
+- de864e7: Keep the terminal anchored where you are reading while an agent streams (#358). Scrolling up during a Codex response could still be dragged back to the live bottom by the next redraw: the flush captured the viewport before writing and restored it immediately after, but xterm parses asynchronously, so at that moment the buffer had not moved yet, the restore compared the anchor against itself and did nothing, and the redraw landed a tick later with nothing left to pull the view back. The restore now runs inside xterm's own write callback, which is the first point at which the redraw's effect exists, and it holds across consecutive and chunked redraws. It is dropped if you switch sessions or a history replay starts before the write parses, since the anchor indexes the buffer it was captured from.
+
## 1.29.1
### Patch Changes
diff --git a/CLAUDE.md b/CLAUDE.md
index b3d765ac..196ad22e 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -77,7 +77,7 @@ When user says "COM":
CI runs `npm run check:lockfile` on every push/PR, so lockfile drift fails the build even if the `version-packages` script is bypassed.
-**Version**: 1.29.1 (must match `package.json`)
+**Version**: 1.30.0 (must match `package.json`)
## Project Overview
diff --git a/package-lock.json b/package-lock.json
index 22edc788..f9640246 100644
--- a/package-lock.json
+++ b/package-lock.json
@@ -1,12 +1,12 @@
{
"name": "aicodeman",
- "version": "1.29.1",
+ "version": "1.30.0",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "aicodeman",
- "version": "1.29.1",
+ "version": "1.30.0",
"hasInstallScript": true,
"license": "MIT",
"workspaces": [
diff --git a/package.json b/package.json
index 287879ba..48a185f6 100644
--- a/package.json
+++ b/package.json
@@ -1,6 +1,6 @@
{
"name": "aicodeman",
- "version": "1.29.1",
+ "version": "1.30.0",
"description": "Mission control for AI coding agents - run 20 autonomous agents with real-time monitoring and session persistence",
"type": "module",
"main": "dist/index.js",
diff --git a/plugins/codeman/.claude-plugin/plugin.json b/plugins/codeman/.claude-plugin/plugin.json
index 8d16d73b..c2def406 100644
--- a/plugins/codeman/.claude-plugin/plugin.json
+++ b/plugins/codeman/.claude-plugin/plugin.json
@@ -1,7 +1,7 @@
{
"name": "codeman",
"description": "Drive Codeman, the self-hosted session manager for AI coding agents, from inside a Claude Code session: spawn worker sessions, prompt them, wait for them, read their answers, clean up. Acts only inside a Codeman-managed session.",
- "version": "1.29.1",
+ "version": "1.30.0",
"author": {
"name": "Ark0N",
"url": "https://github.com/Ark0N"
From 75a028e825d8e38dc55bf42c4b618c92566df114 Mon Sep 17 00:00:00 2001
From: Michael Grundberg
Date: Fri, 18 Sep 2026 16:43:04 +0200
Subject: [PATCH 25/28] fix(terminal): re-take the sticky-scroll baseline after
a replay
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
A capture load now replays its queued tail, and that replay runs through
`batchTerminalWrite`, which samples `_wasAtBottomBeforeWrite` before it queues.
It runs inside `chunkedTerminalWrite`, before that promise resolves, with the
terminal freshly reset and rewritten — so the sample is always true. The caller
then restored the reader's position and the next `flushPendingWrites` scrolled
straight back to the bottom off the latched flag, undoing it. The only thing in
the way was `_hasRecentUserScrollUp()`, a 1500ms window a server-triggered
refresh is usually past.
`_syncStickyScrollBaseline()` re-takes the flag from wherever the viewport now
sits, and the two paths that restore a position call it right after doing so:
`_onSessionNeedsRefresh` and `_maybeRefetchFullHistory`. Those are the paths
#259 and #205 exist for, and they are also where a non-empty queue is most
likely, since a needsRefresh fires when output is flooding. Re-taking rather
than suppressing the sampling: suppressing leaves whatever stale value the flag
held from before the load, which on the full-history re-pull has no reason to
be false. `selectSession` and `_onSessionClearTerminal` deliberately end at the
bottom, so the sampled true is already the truth there and they do not call it.
`_bufferLoadFinishOpts` gains the coverage the CI gate can see: both mux
sources flush, `history` does not, and a payload naming no source does not.
Its only coverage was the browser suite, which CI does not run.
The JSDoc and the changeset now record the one duplicate window this cutoff
cannot close. The server appends output to the byte buffer in the same tick it
emits, but broadcasts on a batch timer — 8ms over WebSocket, 16 to 50ms over
SSE — so a batch pending when `capture-pane` ran leaves the server after the
reply and is replayed although the capture holds it. It is one batch interval
wide against a recovery window spanning the whole chunked write, and closing it
means flushing that batch server side before the capture.
The second browser test asserts its session was created, so a failed create
fails it instead of passing with zero hits.
docs/architecture-invariants.md no longer claims the replay leaves the
queued-event discard window alone. That clause now describes what decides how a
load ends, the baseline rule, the batch window, and the three covering tests.
Co-Authored-By: Claude Opus 5 (1M context)
---
...y-output-that-arrived-after-the-capture.md | 15 +++
docs/architecture-invariants.md | 2 +-
src/web/public/app.js | 20 ++++
src/web/public/terminal-ui.js | 21 ++++
test/capture-load-window.browser.test.ts | 3 +
test/terminal-buffer-flush.test.ts | 98 +++++++++++++++++++
test/terminal-flush-budget.test.ts | 45 +++++++++
7 files changed, 203 insertions(+), 1 deletion(-)
diff --git a/.changeset/fix-replay-output-that-arrived-after-the-capture.md b/.changeset/fix-replay-output-that-arrived-after-the-capture.md
index 03e09189..0cb24af0 100644
--- a/.changeset/fix-replay-output-that-arrived-after-the-capture.md
+++ b/.changeset/fix-replay-output-that-arrived-after-the-capture.md
@@ -34,3 +34,18 @@ takes the flush policy and applies it at its own finish sites.
`_beginBufferLoad` no longer empties the queue when the same load re-enters it,
which it does on every write, because that reset discarded the fetch window
before anything could replay it.
+
+A path that replays its queue and then restores a scroll position re-takes the
+sticky-scroll baseline (`_syncStickyScrollBaseline`). The replay runs with the
+terminal freshly reset, so it reads as sitting at the bottom, and the next flush
+would scroll there and undo the restore. The backpressure refresh and the
+full-history re-pull are the two paths that restore a position, and both are
+ones a reader reaches while scrolled up.
+
+One duplicate window stays open and is not closable from the browser. The server
+appends output to the byte buffer in the same tick it emits, but broadcasts on a
+batch timer, 8ms over WebSocket and 16 to 50ms over SSE. A batch already pending
+when `capture-pane` ran therefore leaves the server after the reply and is
+replayed although the capture holds it. It is one batch interval wide, against a
+recovery window that spans the whole chunked write, and closing it means
+flushing that session's pending batch before taking the capture.
diff --git a/docs/architecture-invariants.md b/docs/architecture-invariants.md
index 20f6bda9..3a0e8733 100644
--- a/docs/architecture-invariants.md
+++ b/docs/architecture-invariants.md
@@ -116,7 +116,7 @@ Tests: `test/docker-hosts.test.ts`, `test/docker-exec-options.test.ts`, `test/do
### Full-scrollback replay
-**Full-scrollback replay** (COD-164/#148, reworked for #205): `GET /api/sessions/:id/terminal?full=1` returns the ENTIRE tmux scrollback (capture-pane `-e -S -` bounded by the configured history limit, explicit `maxBuffer` from the terminal-history config, early byte-cap before normalization, CRLF-normalized for shell panes). On success the capture is returned ALONE (`source='mux-full-history'` — it supersedes the byte buffer; no duplication). The first load of each non-shell TUI session per page requests `full=1` (`_fullHistoryLoaded` Set in app.js — the old one-shot `_initialFullBufferLoad` flag was consumed by whichever tab auto-selected, leaving every other TUI tab one frame of history). Shell sessions instead load a bounded 1 MiB `?tail=` window on every selection and automatic drop recovery: a 100k-line shell capture can be tens of MiB, and automatically parsing it makes tab-switch latency scale with the entire session. Shell full history is explicit-button-only; reaching the top during an ordinary wheel/touch gesture must not reset xterm and replay the multi-megabyte capture on its main thread. Other modes may still re-pull `full=1` at the TOP, and pressing **Load full history** forces the request for any recoverably truncated session (`_maybeRefetchFullHistory`, 4s per-session gesture cooldown, in-flight + tab-switch guards, viewport position held across the replay); Shell full pulls are not retained in the tab cache, so the next switch stays bounded. Chunked replay enqueues 32 KiB pieces across safe yields, appends an xterm parse marker, then releases the live-output gate; output arriving after that release stays ordered behind the snapshot, while the marker callback supplies accurate parse timing without extending the pre-existing queued-event discard window. Live output is separately one-chunk-in-flight: xterm's callback releases each 32/64 KiB write before the next is submitted, keeping the remainder in the app queue where the 128 KiB cap can observe it instead of hiding an unbounded backlog in xterm's private WriteBuffer. While WebSocket owns terminal I/O, parallel SSE terminal/output-recovery events are discarded before JSON parsing; fallback recovery is single-flight per active session so backpressure cannot start overlapping reset+replay cycles. The route exposes capture/prepare totals in `Server-Timing`, while `[TERMINAL-PERF]` separates TTFB, body/JSON, reset+parse and total time for both selection and on-demand full pulls; parse completion is not a browser compositor/GPU paint measurement. The re-pull exists because xterm's buffer is only a WINDOW onto tmux's history and two things shrink it: tmux coalesces bursty output into pane REPAINTS that overwrite rows instead of emitting linefeeds (measured: a 60-line burst added 1 row of browser scrollback and destroyed 34), and a tab switch replays only the visible frame. tmux's own history is intact throughout — the browser just has to ask for it again. On-demand rather than automatic because at a 100k history limit the capture can be megabytes. ⚠️ **The capture ENDS with a cursor move back to the pane's own caret position** (`formatCursorRestore`, from the same `display-message` query the visible-frame path uses). The linear replay otherwise leaves the caret wherever the last character landed — the bottom-most row carrying text, which for an agent CLI is the status line — so the caret sat on the composer's border instead of its input line and every cursor-relative update the CLI sent afterwards was measured from the wrong row, until its next full redraw silently repaired it (that self-repair is why the report read as "it fixes itself as soon as Claude writes a line"). ⚠️ **The move is RELATIVE — up `rows - 1 - cursor_y`, then `\r`, then right `cursor_x` — never `CUP`.** `\x1b[;H` numbers rows from the top of the browser's screen, so it lands correctly only while the browser's row count equals `pane_height`, and nothing guarantees that: `resizeWindow` issues its tmux resize fire-and-forget and returns immediately, so a capture can be taken before a requested resize has applied, and `_onSessionNeedsRefresh` sends no resize at all. Counting up from the last replayed row anchors to the content both ends share. Restoring the cursor makes ROW ALIGNMENT load-bearing on this path: **no transform that can DELETE A LINE may run over a full-history capture**, because every deletion shifts the frame out from under the restored position. Four had accumulated — trailing blank rows stripped by `\n+$`, `stripInkRedrawBloat`, the `CLAUDE_BANNER_PATTERN` trim that cuts everything above the banner, and `LEADING_WHITESPACE_PATTERN` — each correct for a byte stream of successive frames and each wrong for a single rendered frame. ⚠️ **Those skips key on `isFullCapture`, meaning a capture actually came back — never on `?full=1` alone.** When `captureActivePaneBuffer` returns null (ENOBUFS, a timeout, a vanished pane, or a session with no mux at all) the reply falls back to `session.terminalBuffer`, which IS a byte stream and must still be stripped; gating on the query flag returned it whole, and a direct-PTY session takes that path on every first selection rather than only during an outage. ⚠️ A capture holding nothing visible (`hasVisibleContent`) returns `''`, because the caller reads an empty capture as "unavailable" and keeps its byte history — retaining trailing blank rows made an all-blank pane non-empty, which would have replaced real history with a blank screen from the server side, where `_replayWouldShrinkBuffer` cannot see it. ⚠️ **"One line per screen row" holds only where no row was hard-wrapped**: `-J` joins a wrapped row into its logical line (measured: a 100-character line in a 40-column pane captures as 10 lines against a 12-row pane), and the counts reconcile only once the browser xterm re-wraps at the same width — the same assumption `_estimateReplayRows` already documents. Tests: `test/tmux-capture-full-history.test.ts` covers the cursor move, the trim pairing and `hasVisibleContent`; `test/routes/session-routes.test.ts` covers a surviving blank first row, an unstripped byte-history fallback, and an empty capture leaving history intact. ⚠️ **The re-pull must never DOWNGRADE the buffer** (#205 round 2): the same reasoning that makes it a win for a shell pane makes it destructive for a repaint-mode CLI pane, where tmux keeps no history of its own (`history_size≈0` measured for a Claude pane) and the capture is roughly ONE frame while xterm may hold hundreds of rows of replayed frames — `_resetTerminalForReplay()` + rewrite then deletes history mid-scroll ("goes back a bit, repeats blocks, gets worse the further up I go"; measured A/B on a live pane: 341 rows → 42 with the guard off). `_replayWouldShrinkBuffer()` (terminal-ui.js) estimates the capture's rendered rows — escape sequences stripped, `capture-pane -J` re-wrapping accounted for — and the pull is skipped when that is more than one screen short of `buffer.active.length`. The one-screen tolerance matters: both sides are estimates (the buffer length counts trailing blank rows), so only a clear downgrade is refused. A refused session joins `_fullHistoryRepullUseless`, raising its cooldown from 4s to 60s so a hollow pane stops re-fetching megabytes on every scroll-up. Tests: `test/tmux-capture-full-history.test.ts`, `test/tmux-scrollback-eol.test.ts`, `test/terminal-scroll-routing.test.ts`, `test/terminal-flush-budget.test.ts`.
+**Full-scrollback replay** (COD-164/#148, reworked for #205): `GET /api/sessions/:id/terminal?full=1` returns the ENTIRE tmux scrollback (capture-pane `-e -S -` bounded by the configured history limit, explicit `maxBuffer` from the terminal-history config, early byte-cap before normalization, CRLF-normalized for shell panes). On success the capture is returned ALONE (`source='mux-full-history'` — it supersedes the byte buffer; no duplication). The first load of each non-shell TUI session per page requests `full=1` (`_fullHistoryLoaded` Set in app.js — the old one-shot `_initialFullBufferLoad` flag was consumed by whichever tab auto-selected, leaving every other TUI tab one frame of history). Shell sessions instead load a bounded 1 MiB `?tail=` window on every selection and automatic drop recovery: a 100k-line shell capture can be tens of MiB, and automatically parsing it makes tab-switch latency scale with the entire session. Shell full history is explicit-button-only; reaching the top during an ordinary wheel/touch gesture must not reset xterm and replay the multi-megabyte capture on its main thread. Other modes may still re-pull `full=1` at the TOP, and pressing **Load full history** forces the request for any recoverably truncated session (`_maybeRefetchFullHistory`, 4s per-session gesture cooldown, in-flight + tab-switch guards, viewport position held across the replay); Shell full pulls are not retained in the tab cache, so the next switch stays bounded. Chunked replay enqueues 32 KiB pieces across safe yields, appends an xterm parse marker, then releases the live-output gate; output arriving after that release stays ordered behind the snapshot, and the marker callback supplies accurate parse timing. ⚠️ **How the load ENDS depends on where the payload came from**, and `_bufferLoadFinishOpts` (app.js) is the one place that decides it for all four fetch-and-write paths. A payload built from the server's accumulated byte history is current up to the response, so the events queued during the load already appear in it and stay DISCARDED; replaying them would duplicate output, most visibly Ink's cursor-up redraws. A pane capture (`mux-visible` or `mux-full-history`) is current only up to CAPTURE time, so `_finishBufferLoad` replays the queue from the response's own arrival timestamp (`since`) and the pre-capture events stay dropped. ⚠️ **A path that then restores a scroll position must re-take the sticky-scroll baseline** (`_syncStickyScrollBaseline`): the replay runs inside `chunkedTerminalWrite` before its promise resolves, with the terminal freshly reset, so `batchTerminalWrite` samples `_wasAtBottomBeforeWrite` as true and the next `flushPendingWrites` would scroll to the bottom over the restore. ⚠️ The cutoff is a client-side timestamp and the server broadcasts on a batch timer (8ms WebSocket, 16-50ms SSE), so a batch pending when the capture ran arrives after the response and replays although the capture holds it — bounded by one batch interval, and closable only server side by flushing that batch before the capture. Tests for the three: `test/terminal-flush-budget.test.ts` pins which sources flush, `test/terminal-buffer-flush.test.ts` pins the `since` cutoff and the baseline re-take, and `test/capture-load-window.browser.test.ts` drives both against a live server. Live output is separately one-chunk-in-flight: xterm's callback releases each 32/64 KiB write before the next is submitted, keeping the remainder in the app queue where the 128 KiB cap can observe it instead of hiding an unbounded backlog in xterm's private WriteBuffer. While WebSocket owns terminal I/O, parallel SSE terminal/output-recovery events are discarded before JSON parsing; fallback recovery is single-flight per active session so backpressure cannot start overlapping reset+replay cycles. The route exposes capture/prepare totals in `Server-Timing`, while `[TERMINAL-PERF]` separates TTFB, body/JSON, reset+parse and total time for both selection and on-demand full pulls; parse completion is not a browser compositor/GPU paint measurement. The re-pull exists because xterm's buffer is only a WINDOW onto tmux's history and two things shrink it: tmux coalesces bursty output into pane REPAINTS that overwrite rows instead of emitting linefeeds (measured: a 60-line burst added 1 row of browser scrollback and destroyed 34), and a tab switch replays only the visible frame. tmux's own history is intact throughout — the browser just has to ask for it again. On-demand rather than automatic because at a 100k history limit the capture can be megabytes. ⚠️ **The capture ENDS with a cursor move back to the pane's own caret position** (`formatCursorRestore`, from the same `display-message` query the visible-frame path uses). The linear replay otherwise leaves the caret wherever the last character landed — the bottom-most row carrying text, which for an agent CLI is the status line — so the caret sat on the composer's border instead of its input line and every cursor-relative update the CLI sent afterwards was measured from the wrong row, until its next full redraw silently repaired it (that self-repair is why the report read as "it fixes itself as soon as Claude writes a line"). ⚠️ **The move is RELATIVE — up `rows - 1 - cursor_y`, then `\r`, then right `cursor_x` — never `CUP`.** `\x1b[;H` numbers rows from the top of the browser's screen, so it lands correctly only while the browser's row count equals `pane_height`, and nothing guarantees that: `resizeWindow` issues its tmux resize fire-and-forget and returns immediately, so a capture can be taken before a requested resize has applied, and `_onSessionNeedsRefresh` sends no resize at all. Counting up from the last replayed row anchors to the content both ends share. Restoring the cursor makes ROW ALIGNMENT load-bearing on this path: **no transform that can DELETE A LINE may run over a full-history capture**, because every deletion shifts the frame out from under the restored position. Four had accumulated — trailing blank rows stripped by `\n+$`, `stripInkRedrawBloat`, the `CLAUDE_BANNER_PATTERN` trim that cuts everything above the banner, and `LEADING_WHITESPACE_PATTERN` — each correct for a byte stream of successive frames and each wrong for a single rendered frame. ⚠️ **Those skips key on `isFullCapture`, meaning a capture actually came back — never on `?full=1` alone.** When `captureActivePaneBuffer` returns null (ENOBUFS, a timeout, a vanished pane, or a session with no mux at all) the reply falls back to `session.terminalBuffer`, which IS a byte stream and must still be stripped; gating on the query flag returned it whole, and a direct-PTY session takes that path on every first selection rather than only during an outage. ⚠️ A capture holding nothing visible (`hasVisibleContent`) returns `''`, because the caller reads an empty capture as "unavailable" and keeps its byte history — retaining trailing blank rows made an all-blank pane non-empty, which would have replaced real history with a blank screen from the server side, where `_replayWouldShrinkBuffer` cannot see it. ⚠️ **"One line per screen row" holds only where no row was hard-wrapped**: `-J` joins a wrapped row into its logical line (measured: a 100-character line in a 40-column pane captures as 10 lines against a 12-row pane), and the counts reconcile only once the browser xterm re-wraps at the same width — the same assumption `_estimateReplayRows` already documents. Tests: `test/tmux-capture-full-history.test.ts` covers the cursor move, the trim pairing and `hasVisibleContent`; `test/routes/session-routes.test.ts` covers a surviving blank first row, an unstripped byte-history fallback, and an empty capture leaving history intact. ⚠️ **The re-pull must never DOWNGRADE the buffer** (#205 round 2): the same reasoning that makes it a win for a shell pane makes it destructive for a repaint-mode CLI pane, where tmux keeps no history of its own (`history_size≈0` measured for a Claude pane) and the capture is roughly ONE frame while xterm may hold hundreds of rows of replayed frames — `_resetTerminalForReplay()` + rewrite then deletes history mid-scroll ("goes back a bit, repeats blocks, gets worse the further up I go"; measured A/B on a live pane: 341 rows → 42 with the guard off). `_replayWouldShrinkBuffer()` (terminal-ui.js) estimates the capture's rendered rows — escape sequences stripped, `capture-pane -J` re-wrapping accounted for — and the pull is skipped when that is more than one screen short of `buffer.active.length`. The one-screen tolerance matters: both sides are estimates (the buffer length counts trailing blank rows), so only a clear downgrade is refused. A refused session joins `_fullHistoryRepullUseless`, raising its cooldown from 4s to 60s so a hollow pane stops re-fetching megabytes on every scroll-up. Tests: `test/tmux-capture-full-history.test.ts`, `test/tmux-scrollback-eol.test.ts`, `test/terminal-scroll-routing.test.ts`, `test/terminal-flush-budget.test.ts`.
### Terminal scrollback: strip flavors and wheel/touch forwarding
diff --git a/src/web/public/app.js b/src/web/public/app.js
index ab303da5..e96e78b7 100644
--- a/src/web/public/app.js
+++ b/src/web/public/app.js
@@ -1921,6 +1921,16 @@ class CodemanApp {
* the moment the response arrived, compared only against other client-side
* readings, so there is no clock skew to worry about.
*
+ * What this cutoff does NOT cover: the server appends output to the byte
+ * buffer and emits it in the same tick, but it BROADCASTS on a batch timer —
+ * 8ms over WebSocket, 16 to 50ms over SSE. The terminal route runs
+ * synchronously from `capture-pane` to its return, so a batch that was
+ * already pending when the capture ran leaves the server after the reply,
+ * arrives after `headersReceivedAt`, and is replayed although the capture
+ * holds it. The duplicate is one batch interval wide, against a recovery
+ * window that spans the whole chunked write. Closing it belongs on the
+ * server: flush that session's pending batch before taking the capture.
+ *
* @param {{source?: string}} payload - The parsed `data` of a terminal response.
* @param {number} headersReceivedAt - When that response reached this client.
* @returns {{flushQueued: boolean, since: number}} Options for `_finishBufferLoad`.
@@ -2557,6 +2567,10 @@ class CodemanApp {
});
if (target === null || typeof this.terminal.scrollToLine !== 'function') this.terminal.scrollToBottom();
else this.terminal.scrollToLine(target);
+ // The load's own replay sampled the sticky-scroll baseline while the
+ // terminal sat at the bottom of a just-rewritten buffer, so the next
+ // flush would scroll back down and undo the restore above.
+ this._syncStickyScrollBaseline();
// Re-position local echo overlay at new prompt location
this._localEchoOverlay?.rerender();
// Resize PTY to match actual browser dimensions (critical for OpenCode
@@ -5842,6 +5856,12 @@ class CodemanApp {
const delta = parsedBufferLength - rowsBefore;
if (delta > 0) this.terminal.scrollToLine(delta);
else this.terminal.scrollToTop();
+ // The load's own replay sampled the sticky-scroll baseline while the
+ // terminal sat at the bottom of a just-rewritten buffer, so the next
+ // flush would scroll back down and undo the restore above. This path is
+ // reached only from a scroll-up gesture, so being dragged down is the
+ // exact opposite of what the user asked for.
+ this._syncStickyScrollBaseline();
timing.totalMs = performance.now() - requestStartedAt;
this._recordTerminalLoadTiming(timing);
} catch {
diff --git a/src/web/public/terminal-ui.js b/src/web/public/terminal-ui.js
index 1a265d61..b56f5510 100644
--- a/src/web/public/terminal-ui.js
+++ b/src/web/public/terminal-ui.js
@@ -3131,6 +3131,27 @@ Object.assign(CodemanApp.prototype, {
return buffer.viewportY >= buffer.baseY - 2;
},
+ /**
+ * Re-take the sticky-scroll baseline from where the viewport now sits.
+ *
+ * `batchTerminalWrite` samples `_wasAtBottomBeforeWrite` before it queues
+ * data, and `flushPendingWrites` scrolls to the bottom off that sample. A
+ * buffer load that replays its queue samples at the worst possible moment:
+ * `_finishBufferLoad` runs inside `chunkedTerminalWrite`, before its promise
+ * resolves, with the terminal freshly reset and rewritten, so the sample is
+ * always true. A caller that then restores the reader's position would have
+ * that restore undone by the next flush.
+ *
+ * Every caller that scrolls the viewport somewhere other than the bottom
+ * after a load must call this, so the baseline describes the position the
+ * caller chose. `selectSession` and `_onSessionClearTerminal` deliberately
+ * end at the bottom, so for them the sampled true is already the truth and
+ * they do not call it.
+ */
+ _syncStickyScrollBaseline() {
+ this._wasAtBottomBeforeWrite = this.isTerminalAtBottom();
+ },
+
// Record manual scroll gestures so sticky-scroll can give an upward scroll a
// short grace window (see _hasRecentUserScrollUp). A downward scroll that
// lands back at the bottom clears the suppression immediately.
diff --git a/test/capture-load-window.browser.test.ts b/test/capture-load-window.browser.test.ts
index 9f1ee810..8cea77ef 100644
--- a/test/capture-load-window.browser.test.ts
+++ b/test/capture-load-window.browser.test.ts
@@ -165,6 +165,9 @@ describe('output emitted during a capture load', () => {
context = await browser.newContext({ viewport: { width: 1280, height: 800 } });
page = await context.newPage();
const sessionId = await openSession(page);
+ // Without this, a failed create passes the zero-hit assertion below
+ // vacuously — nothing was loaded, so nothing was replayed.
+ expect(sessionId).toBeTruthy();
expect(await runLoad(page, sessionId, 'history')).toBe(0);
diff --git a/test/terminal-buffer-flush.test.ts b/test/terminal-buffer-flush.test.ts
index 0500a3e6..36f652ef 100644
--- a/test/terminal-buffer-flush.test.ts
+++ b/test/terminal-buffer-flush.test.ts
@@ -84,6 +84,50 @@ function makeApp() {
return { app, writes };
}
+/**
+ * A stub carrying the REAL `batchTerminalWrite` on top of the real begin/finish
+ * methods, so a replay samples the sticky-scroll baseline exactly as it does in
+ * the browser. The terminal is a fake whose `buffer.active` the test moves by
+ * hand, which is what a caller's `scrollToLine` does to a real one.
+ */
+function makeScrollApp() {
+ const buffer = { viewportY: 0, baseY: 100 };
+ const app = {
+ buffer,
+ terminal: { buffer: { active: buffer } },
+ sessions: new Map(),
+ activeSessionId: null,
+ pendingWrites: [] as string[],
+ writeFrameScheduled: false,
+ _wasAtBottomBeforeWrite: false,
+ _bufferLoadSeq: 0,
+ _bufferLoadOwner: null as string | null,
+ _isLoadingBuffer: false,
+ _loadBufferQueue: null as { at: number; data: string }[] | null,
+ _scheduleTerminalWriteFlush: vi.fn(),
+ batchTerminalWrite: mixin.batchTerminalWrite as (data: string) => void,
+ isTerminalAtBottom: mixin.isTerminalAtBottom as () => boolean,
+ _syncStickyScrollBaseline: mixin._syncStickyScrollBaseline as () => void,
+ _beginBufferLoad: mixin._beginBufferLoad as BufferLoadApp['_beginBufferLoad'],
+ _finishBufferLoad: mixin._finishBufferLoad as BufferLoadApp['_finishBufferLoad'],
+ };
+ return app;
+}
+
+/**
+ * Slice one class method out of app.js, from its header to the next method's.
+ *
+ * Bounding the slice matters: the two methods checked below are not followed by
+ * a JSDoc block, so a scan for the next comment would run on into unrelated
+ * code and match its scroll calls instead of theirs.
+ */
+function methodBody(source: string, method: string): string {
+ const start = source.search(new RegExp(`^ {2}(?:async )?${method}\\(`, 'm'));
+ expect(start, `${method} not found in app.js`).toBeGreaterThan(-1);
+ const next = /^ {2}(?:async )?[A-Za-z_$][\w$]*\(/m.exec(source.slice(start + 1));
+ return next ? source.slice(start, start + 1 + next.index) : source.slice(start);
+}
+
/**
* Simulate a live SSE event arriving while a buffer load is in progress.
* Mirrors batchTerminalWrite's queue branch, which stamps each entry with its
@@ -241,6 +285,60 @@ describe('buffer-load flush (COD-144)', () => {
expect(writes).toEqual(['belongs-to-this-load']);
});
+ // ── The sticky-scroll baseline across a replay ──
+ //
+ // `batchTerminalWrite` samples `_wasAtBottomBeforeWrite` before queueing, and
+ // `flushPendingWrites` scrolls to the bottom off that sample. The replay runs
+ // inside `chunkedTerminalWrite` before its promise resolves, with the terminal
+ // freshly reset and rewritten, so the sample is always true. A caller that
+ // then restores the reader's position would have that restore undone.
+
+ it('the replay latches the baseline true, and the viewport restore re-takes it', () => {
+ const app = makeScrollApp();
+ const owner = app._beginBufferLoad('load-scroll');
+ pushWhileLoading(app as unknown as BufferLoadApp, 'output-after-the-capture', 100);
+
+ // The load ends with the terminal reset and rewritten, so it reads as bottom.
+ app.buffer.viewportY = app.buffer.baseY;
+ app._finishBufferLoad(owner, { flushQueued: true, since: 0 });
+ expect(app._wasAtBottomBeforeWrite).toBe(true);
+
+ // The caller now puts the reader back where they were reading.
+ app.buffer.viewportY = 40;
+ app._syncStickyScrollBaseline();
+
+ // The next flush must leave them there.
+ expect(app._wasAtBottomBeforeWrite).toBe(false);
+ });
+
+ it('a restore that lands back at the bottom keeps sticky scroll armed', () => {
+ const app = makeScrollApp();
+ const owner = app._beginBufferLoad('load-scroll-bottom');
+ pushWhileLoading(app as unknown as BufferLoadApp, 'output-after-the-capture', 100);
+
+ app.buffer.viewportY = app.buffer.baseY;
+ app._finishBufferLoad(owner, { flushQueued: true, since: 0 });
+ app._syncStickyScrollBaseline();
+
+ // A reader who was already at the bottom still wants to be carried along.
+ expect(app._wasAtBottomBeforeWrite).toBe(true);
+ });
+
+ it('both callers that restore a scroll position re-take the baseline', () => {
+ // The wiring lives in app.js, outside this file's vm harness. Without it the
+ // two methods below restore the viewport and the next flush undoes it.
+ const source = readFileSync(resolve(import.meta.dirname, '../src/web/public/app.js'), 'utf8');
+
+ for (const method of ['_onSessionNeedsRefresh', '_maybeRefetchFullHistory']) {
+ const body = methodBody(source, method);
+ const restoreAt = body.lastIndexOf('scrollToLine(');
+ const syncAt = body.indexOf('this._syncStickyScrollBaseline()');
+ expect(restoreAt, `${method} no longer restores a scroll position`).toBeGreaterThan(-1);
+ expect(syncAt, `${method} never re-takes the baseline`).toBeGreaterThan(-1);
+ expect(syncAt, `${method} re-takes the baseline before its restore`).toBeGreaterThan(restoreAt);
+ }
+ });
+
it('empty queue + flushQueued is a no-op (no throw, no writes)', () => {
const { app, writes } = makeApp();
const owner = app._beginBufferLoad('load-empty');
diff --git a/test/terminal-flush-budget.test.ts b/test/terminal-flush-budget.test.ts
index 514e93da..aa103661 100644
--- a/test/terminal-flush-budget.test.ts
+++ b/test/terminal-flush-budget.test.ts
@@ -322,6 +322,51 @@ describe('terminal flush budget', () => {
expect(app._bufferLoadOwner).toBe(null);
});
+ // ── Which payloads end their load by replaying the queue ──
+ //
+ // A pane capture is current only up to capture time, so the tail that arrived
+ // after the response exists nowhere else and has to be replayed. The server's
+ // accumulated byte history is current up to the response, so replaying on top
+ // of it would duplicate output. `_bufferLoadFinishOpts` is the one place that
+ // decides this, for all four paths that fetch a terminal buffer and write it.
+
+ it('replays the tail for a visible-pane capture', () => {
+ const { CodemanApp } = loadAppHarness();
+ const app = Object.create(CodemanApp.prototype) as any;
+
+ expect(app._bufferLoadFinishOpts({ source: 'mux-visible' }, 1234)).toEqual({
+ flushQueued: true,
+ since: 1234,
+ });
+ });
+
+ it('replays the tail for a full-history capture', () => {
+ const { CodemanApp } = loadAppHarness();
+ const app = Object.create(CodemanApp.prototype) as any;
+
+ expect(app._bufferLoadFinishOpts({ source: 'mux-full-history' }, 1234)).toEqual({
+ flushQueued: true,
+ since: 1234,
+ });
+ });
+
+ it('discards the queue for the accumulated byte history', () => {
+ const { CodemanApp } = loadAppHarness();
+ const app = Object.create(CodemanApp.prototype) as any;
+
+ expect(app._bufferLoadFinishOpts({ source: 'history' }, 1234).flushQueued).toBe(false);
+ });
+
+ it('discards the queue for a payload that names no source', () => {
+ // Fails toward the safe answer: a duplicated Ink redraw corrupts the screen,
+ // while a dropped tail is repaired by the CLI's next full repaint.
+ const { CodemanApp } = loadAppHarness();
+ const app = Object.create(CodemanApp.prototype) as any;
+
+ expect(app._bufferLoadFinishOpts({}, 1234).flushQueued).toBe(false);
+ expect(app._bufferLoadFinishOpts(undefined, 1234).flushQueued).toBe(false);
+ });
+
it('does not snap back to bottom during Codex Working redraws right after the user scrolls up', () => {
const { app } = loadTerminalUiHarness('codex');
const scrollToBottom = vi.fn();
From cfd771d1d8121e80791b8cc9cf148853c5b845d1 Mon Sep 17 00:00:00 2001
From: Michael Grundberg
Date: Fri, 18 Sep 2026 18:44:04 +0200
Subject: [PATCH 26/28] test(terminal): pin all four buffer-load paths to the
shared flush helper
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The first version of this fix decided the flush policy in `selectSession`
alone, and a later pass found it still covering one path of four. Nothing in
the CI gate stops a fifth path, or an inlined `{ flushQueued: true }`, from
splitting that policy up again — the browser suite that would notice is
excluded from `npm test`.
A static scan over `selectSession`, `_onSessionNeedsRefresh`,
`_onSessionClearTerminal` and `_maybeRefetchFullHistory` asserts each one asks
`_bufferLoadFinishOpts`, reusing the `methodBody` slice the sticky-scroll guard
already needed. Verified by inlining the policy back into
`_onSessionClearTerminal`, which fails it by name.
Co-Authored-By: Claude Opus 5 (1M context)
---
test/terminal-buffer-flush.test.ts | 20 ++++++++++++++++++++
1 file changed, 20 insertions(+)
diff --git a/test/terminal-buffer-flush.test.ts b/test/terminal-buffer-flush.test.ts
index 36f652ef..9bc61771 100644
--- a/test/terminal-buffer-flush.test.ts
+++ b/test/terminal-buffer-flush.test.ts
@@ -339,6 +339,26 @@ describe('buffer-load flush (COD-144)', () => {
}
});
+ it('every path that fetches a terminal buffer and writes it asks the shared helper', () => {
+ // Drift guard. The first version of this fix covered one of the four paths,
+ // and a later pass found it still covering one of four. Nothing else in the
+ // gate stops a fifth path, or an inlined `{ flushQueued: true }`, from
+ // splitting the policy up again; the browser suite that would notice does
+ // not run in CI.
+ const source = readFileSync(resolve(import.meta.dirname, '../src/web/public/app.js'), 'utf8');
+
+ for (const method of [
+ 'selectSession',
+ '_onSessionNeedsRefresh',
+ '_onSessionClearTerminal',
+ '_maybeRefetchFullHistory',
+ ]) {
+ expect(methodBody(source, method), `${method} decides the flush policy itself`).toContain(
+ 'this._bufferLoadFinishOpts('
+ );
+ }
+ });
+
it('empty queue + flushQueued is a no-op (no throw, no writes)', () => {
const { app, writes } = makeApp();
const owner = app._beginBufferLoad('load-empty');
From 3730bc7df5279c073b23f82f2eb86f8913abe993 Mon Sep 17 00:00:00 2001
From: Michael Grundberg
Date: Fri, 18 Sep 2026 18:56:15 +0200
Subject: [PATCH 27/28] docs(terminal): correct what selectSession does with
the viewport
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The JSDoc on `_syncStickyScrollBaseline` said `selectSession` deliberately ends
at the bottom, so the baseline the replay samples is already true there. It
does not. `selectSession` calls `scrollToBottom()` after the write and then
ends at `scrollToLastNonEmptyLine()` (app.js:6512), which targets
`lastNonEmptyLine - rows + 2` and therefore parks ABOVE `baseY` whenever the
replayed frame keeps trailing blank rows — which a full capture does on
purpose, since no transform that can delete a line may run over one.
Its baseline really is a stale true. What covers it is the sticky snap itself:
since de864e7d that snap fires only when the flush found the viewport already
at the bottom (`preserveViewportY === null`), which a parked selectSession
viewport is not. That commit landed on master after this branch was cut, so
the guard arrives with the merge rather than being present here.
`_onSessionClearTerminal` is unchanged in the comment and was correct: it
resets and rewrites with no scroll afterwards, so it does end at the bottom.
Comment only; no behaviour change.
Co-Authored-By: Claude Opus 5 (1M context)
---
src/web/public/terminal-ui.js | 21 ++++++++++++++++-----
1 file changed, 16 insertions(+), 5 deletions(-)
diff --git a/src/web/public/terminal-ui.js b/src/web/public/terminal-ui.js
index b56f5510..9b56d0aa 100644
--- a/src/web/public/terminal-ui.js
+++ b/src/web/public/terminal-ui.js
@@ -3142,11 +3142,22 @@ Object.assign(CodemanApp.prototype, {
* always true. A caller that then restores the reader's position would have
* that restore undone by the next flush.
*
- * Every caller that scrolls the viewport somewhere other than the bottom
- * after a load must call this, so the baseline describes the position the
- * caller chose. `selectSession` and `_onSessionClearTerminal` deliberately
- * end at the bottom, so for them the sampled true is already the truth and
- * they do not call it.
+ * `_onSessionNeedsRefresh` and `_maybeRefetchFullHistory` restore a position
+ * and both call this, so their baseline describes the position they chose.
+ *
+ * The other two load paths do not call it, for different reasons.
+ * `_onSessionClearTerminal` resets and rewrites with no scroll afterwards,
+ * so the sampled true is already the truth there. `selectSession` does NOT
+ * end at the bottom, whatever its `scrollToBottom()` after the write
+ * suggests: it ends at `scrollToLastNonEmptyLine()`, which targets
+ * `lastNonEmptyLine - rows + 2` and therefore parks ABOVE `baseY` whenever
+ * the replayed frame keeps trailing blank rows, which a full capture does on
+ * purpose. Its baseline is a stale true. What decides whether that matters
+ * is the sticky snap in `flushPendingWrites`, and since de864e7d that snap
+ * fires only when the flush found the viewport already at the bottom
+ * (`preserveViewportY === null`), which a parked selectSession viewport is
+ * not. Do not read the absent call here as a claim that selectSession lands
+ * at the bottom.
*/
_syncStickyScrollBaseline() {
this._wasAtBottomBeforeWrite = this.isTerminalAtBottom();
From 3cdb4bf42ecb2028b9a2c5aaa2b181bac7343630 Mon Sep 17 00:00:00 2001
From: Codeman maintainer
Date: Fri, 18 Sep 2026 21:38:40 +0200
Subject: [PATCH 28/28] docs(terminal): the merge-time notes promised on #436
The four edits the review said would be folded in at merge, none of them
code: the changeset becomes one user-facing paragraph, since it is what
CHANGELOG.md and the release notes print; the `_bufferLoadFinishOpts` comment
now names the second contributor to the duplicate window (`captureActivePaneBuffer`
is `execSync`, so anything painted into the pane before the server read it is
in the capture and is broadcast after the reply) and says why a `history`
payload keeps the pre-existing discard when its exposure is the same; the
`_finishBufferLoad` doc block moves from above `_beginBufferLoad` onto the
function it documents; and the test file's header describes both rules the
file now pins instead of only COD-144.
Co-Authored-By: Claude Fable 5.1
---
...y-output-that-arrived-after-the-capture.md | 48 +------------------
src/web/public/app.js | 37 +++++++++-----
src/web/public/terminal-ui.js | 39 +++++++++------
test/terminal-buffer-flush.test.ts | 39 +++++++++------
4 files changed, 75 insertions(+), 88 deletions(-)
diff --git a/.changeset/fix-replay-output-that-arrived-after-the-capture.md b/.changeset/fix-replay-output-that-arrived-after-the-capture.md
index 0cb24af0..023623ef 100644
--- a/.changeset/fix-replay-output-that-arrived-after-the-capture.md
+++ b/.changeset/fix-replay-output-that-arrived-after-the-capture.md
@@ -2,50 +2,4 @@
"aicodeman": patch
---
-fix(terminal): keep the output a pane capture could not contain
-
-Live terminal events are queued while a buffer load runs, and the load discards
-that queue when it ends. That is right when the loaded buffer is the server's
-accumulated byte history: the route appends to that history right up to the
-moment it serializes the response, so the queued events already appear in it and
-replaying them would duplicate output.
-
-A tmux pane capture is a photograph, current only as of the instant
-`capture-pane` ran. Output printed afterwards was queued and then dropped, with
-nothing scheduling a re-fetch, and the CLI's next partial redraw landed on a
-frame the terminal never received. A `?full=1` load returns the capture alone,
-so it lost everything from the capture to the end of the chunked write. A
-`?tail=` load carries the byte history in front of the capture, so it lost
-everything from the response to the end of that write. A shell session shows
-this most plainly, because its output is linear and nothing repaints it.
-
-Queue entries now carry their arrival time, and `_finishBufferLoad` takes a
-`since` cutoff so a capture load replays exactly the tail that arrived after the
-response headers. All four paths that fetch a terminal buffer and write it use
-the same rule, through one shared `_bufferLoadFinishOpts` helper: selecting a
-session, the backpressure refresh, the clear-terminal reload, and the
-full-history re-pull. The backpressure refresh matters most, because it exists
-to restore output the client already dropped once and could drop more while
-doing it.
-
-Two things had to change for that tail to still exist when the load ends.
-`chunkedTerminalWrite` is what ends the load for any non-empty buffer, so it
-takes the flush policy and applies it at its own finish sites.
-`_beginBufferLoad` no longer empties the queue when the same load re-enters it,
-which it does on every write, because that reset discarded the fetch window
-before anything could replay it.
-
-A path that replays its queue and then restores a scroll position re-takes the
-sticky-scroll baseline (`_syncStickyScrollBaseline`). The replay runs with the
-terminal freshly reset, so it reads as sitting at the bottom, and the next flush
-would scroll there and undo the restore. The backpressure refresh and the
-full-history re-pull are the two paths that restore a position, and both are
-ones a reader reaches while scrolled up.
-
-One duplicate window stays open and is not closable from the browser. The server
-appends output to the byte buffer in the same tick it emits, but broadcasts on a
-batch timer, 8ms over WebSocket and 16 to 50ms over SSE. A batch already pending
-when `capture-pane` ran therefore leaves the server after the reply and is
-replayed although the capture holds it. It is one batch interval wide, against a
-recovery window that spans the whole chunked write, and closing it means
-flushing that session's pending batch before taking the capture.
+fix(terminal): keep the output a pane capture could not contain. Opening a session, a backpressure refresh, a clear-terminal reload and a full-history re-pull all load the screen from a tmux pane capture, and anything the CLI printed between that capture and the end of the load used to be dropped, so its next partial redraw landed on a frame the terminal had never seen: missing or garbled output right after a tab switch or a refresh, plainest in a shell session. Each load now replays exactly the output that arrived after the capture, through one shared rule for all four paths, and a refresh that restores your scroll position no longer snaps back to the bottom afterwards.
diff --git a/src/web/public/app.js b/src/web/public/app.js
index 88fb115d..bb40152d 100644
--- a/src/web/public/app.js
+++ b/src/web/public/app.js
@@ -1917,23 +1917,36 @@ class CodemanApp {
* browser after the response headers can already be in it. Such a load
* replays exactly that tail; discarding it drops the CLI's output for the
* rest of the load window, and its next partial redraw then lands on a frame
- * the terminal never received. A payload built from the server's accumulated
- * byte history needs the opposite: that history is current up to the
- * response, so replaying the queue on top of it would duplicate output.
+ * the terminal never received.
+ *
+ * A `history` payload is the server's byte buffer alone: the direct-PTY
+ * fallback, or a mux pane whose capture came back empty. The route reads
+ * that buffer in the same synchronous tick it takes the capture, so it is
+ * current up to the route's own read and no further, which is the same
+ * exposure. It deliberately keeps the pre-existing discard all the same:
+ * both cases are rare, neither has been measured, and a duplicated Ink
+ * redraw is more visible than a few milliseconds of missing output.
+ * `capturedFromMux` below is the one line to widen if either turns out to
+ * matter.
*
* `headersReceivedAt` is the caller's own `performance.now()` reading from
* the moment the response arrived, compared only against other client-side
* readings, so there is no clock skew to worry about.
*
- * What this cutoff does NOT cover: the server appends output to the byte
- * buffer and emits it in the same tick, but it BROADCASTS on a batch timer —
- * 8ms over WebSocket, 16 to 50ms over SSE. The terminal route runs
- * synchronously from `capture-pane` to its return, so a batch that was
- * already pending when the capture ran leaves the server after the reply,
- * arrives after `headersReceivedAt`, and is replayed although the capture
- * holds it. The duplicate is one batch interval wide, against a recovery
- * window that spans the whole chunked write. Closing it belongs on the
- * server: flush that session's pending batch before taking the capture.
+ * What this cutoff does NOT cover, and there are two contributors. The
+ * server appends output to the byte buffer and emits it in the same tick,
+ * but BROADCASTS on a batch timer (8ms over WebSocket, 16 to 50ms over SSE),
+ * and the terminal route runs synchronously from `capture-pane` to its
+ * return, so a batch already pending when the capture ran leaves the server
+ * after the reply, arrives after `headersReceivedAt`, and is replayed
+ * although the capture holds it. Separately, `captureActivePaneBuffer` is
+ * `execSync`, which blocks the event loop for the whole capture: anything
+ * tmux had already painted into the pane that the server had not yet read
+ * from the attach PTY is in the capture too, is broadcast only after the
+ * reply, and replays the same way. The duplicate is one batch interval plus
+ * one capture wide, against a recovery window that spans the whole chunked
+ * write. Closing it belongs on the server: flush that session's pending
+ * batch before taking the capture.
*
* @param {{source?: string}} payload - The parsed `data` of a terminal response.
* @param {number} headersReceivedAt - When that response reached this client.
diff --git a/src/web/public/terminal-ui.js b/src/web/public/terminal-ui.js
index 5c1af2e7..c767ad36 100644
--- a/src/web/public/terminal-ui.js
+++ b/src/web/public/terminal-ui.js
@@ -3924,6 +3924,30 @@ Object.assign(CodemanApp.prototype, {
});
},
+ /**
+ * Open a buffer load: live terminal events are queued from here until
+ * `_finishBufferLoad` decides what to do with them. Returns the load token the
+ * finish call must present; a stale token makes that call a no-op.
+ *
+ * @param {string} [owner] Reuse an existing token to re-enter the same load
+ * (see below); omit it to start a new one.
+ * @returns {string} The load token.
+ */
+ _beginBufferLoad(owner) {
+ if (this._bufferLoadSeq === undefined) this._bufferLoadSeq = 0;
+ const loadOwner = owner === undefined ? `buffer-${++this._bufferLoadSeq}` : owner;
+ // `selectSession` opens the load before its fetch, and `chunkedTerminalWrite`
+ // opens it again under the SAME owner when it starts writing. Resetting the
+ // queue on that second call would throw away everything that arrived during
+ // the fetch, which on the capture path is output no buffer holds. Re-entering
+ // one load keeps its queue; a genuinely new load still starts empty.
+ const reentering = this._bufferLoadOwner === loadOwner && Array.isArray(this._loadBufferQueue);
+ this._bufferLoadOwner = loadOwner;
+ this._isLoadingBuffer = true;
+ if (!reentering) this._loadBufferQueue = [];
+ return loadOwner;
+ },
+
/**
* Complete a buffer load: unblock live SSE writes.
* Called when chunkedTerminalWrite finishes (or is skipped for empty buffers).
@@ -3960,21 +3984,6 @@ Object.assign(CodemanApp.prototype, {
* is true, replay queued events whose arrival timestamp is at or after
* `since` (default 0, meaning the whole queue).
*/
- _beginBufferLoad(owner) {
- if (this._bufferLoadSeq === undefined) this._bufferLoadSeq = 0;
- const loadOwner = owner === undefined ? `buffer-${++this._bufferLoadSeq}` : owner;
- // `selectSession` opens the load before its fetch, and `chunkedTerminalWrite`
- // opens it again under the SAME owner when it starts writing. Resetting the
- // queue on that second call would throw away everything that arrived during
- // the fetch, which on the capture path is output no buffer holds. Re-entering
- // one load keeps its queue; a genuinely new load still starts empty.
- const reentering = this._bufferLoadOwner === loadOwner && Array.isArray(this._loadBufferQueue);
- this._bufferLoadOwner = loadOwner;
- this._isLoadingBuffer = true;
- if (!reentering) this._loadBufferQueue = [];
- return loadOwner;
- },
-
_finishBufferLoad(owner, opts) {
if (owner !== undefined && this._bufferLoadOwner !== owner) {
return false;
diff --git a/test/terminal-buffer-flush.test.ts b/test/terminal-buffer-flush.test.ts
index 9bc61771..bacd2923 100644
--- a/test/terminal-buffer-flush.test.ts
+++ b/test/terminal-buffer-flush.test.ts
@@ -1,20 +1,31 @@
/**
- * @fileoverview Regression tests for the buffer-load flush path (COD-144).
+ * @fileoverview Regression tests for the buffer-load flush path: what becomes of
+ * the live terminal events queued while a buffer load runs, once the load ends.
*
- * Bug: newly launched Shell sessions rendered BLANK until a tab-switch. The
- * buffer-load path (`selectSession` → `_beginBufferLoad`/`_finishBufferLoad`)
- * QUEUES live SSE terminal events while `_isLoadingBuffer` is true, then on
- * completion DISCARDS the queue (`_loadBufferQueue = null`). That de-dup is
- * correct for an established session (the fetched buffer already contains the
- * queued output, so replaying it would duplicate Ink redraws). But for a
- * brand-new shell the fetch resolves BEFORE the PTY emits its prompt — the
- * fetched buffer is empty and the prompt arrives only as a queued event, which
- * then gets discarded → blank terminal.
+ * Two rules, each from a real bug.
*
- * Fix: `_finishBufferLoad(owner, { flushQueued })` REPLAYS the queued events
- * through `batchTerminalWrite()` (after `_isLoadingBuffer` is cleared, so they
- * write through normally) ONLY when the load painted nothing. The default path
- * (no opts) still discards, preserving de-dup for established sessions.
+ * COD-144: newly launched Shell sessions rendered BLANK until a tab-switch. The
+ * load path (`selectSession` → `_beginBufferLoad`/`_finishBufferLoad`) queues
+ * live events while `_isLoadingBuffer` is true and used to DISCARD the queue on
+ * completion. Right for a buffer built from the server's byte history (the
+ * queued output is already in it, so replaying it duplicates Ink redraws),
+ * wrong for a brand-new shell whose fetch resolves BEFORE the PTY emits its
+ * prompt: the prompt arrived only as a queued event and was thrown away. A
+ * caller that knows the load painted nothing passes `{ flushQueued: true }`
+ * and the queue is REPLAYED through `batchTerminalWrite()` after
+ * `_isLoadingBuffer` is cleared, so the events write through normally.
+ *
+ * #436: a tmux pane capture is current only as of the instant `capture-pane`
+ * ran, so everything the CLI printed between the capture and the end of the
+ * chunked write was queued and dropped, and its next partial redraw landed on
+ * a frame the terminal never received. Queue entries now carry their arrival
+ * time and `_finishBufferLoad` takes a `since` cutoff, so a capture load
+ * replays exactly the tail that arrived after the response headers. All four
+ * fetch-and-write paths take that policy from one helper,
+ * `_bufferLoadFinishOpts`, and a static scan below pins each of them to it,
+ * because the same fix had already been written into one path out of four,
+ * twice. A path that replays and then restores a scroll position re-takes the
+ * sticky-scroll baseline (`_syncStickyScrollBaseline`), pinned the same way.
*
* Loaded via `vm` with a stubbed context (no jsdom — jsdom is broken on this
* box; see connection-indicator.test.ts). We extract the REAL