From 268e4819ff21dce20064c307ac9b7c5dacfa955d Mon Sep 17 00:00:00 2001 From: codeman-local Date: Thu, 3 Sep 2026 18:14:29 +0800 Subject: [PATCH 01/28] feat: auto-name sessions from first prompt --- docs/wiki/The-Dashboard.md | 8 +++ src/session-auto-name.ts | 101 +++++++++++++++++++++++++++ src/session.ts | 44 ++++++++++-- src/types/session.ts | 5 ++ src/web/routes/session-routes.ts | 2 + src/web/server.ts | 2 + src/web/session-listener-wiring.ts | 18 ++++- test/session-listener-wiring.test.ts | 27 +++++++ test/session-submit-anchor.test.ts | 29 ++++++++ 9 files changed, 229 insertions(+), 7 deletions(-) create mode 100644 src/session-auto-name.ts diff --git a/docs/wiki/The-Dashboard.md b/docs/wiki/The-Dashboard.md index d5130869..f6c2d171 100644 --- a/docs/wiki/The-Dashboard.md +++ b/docs/wiki/The-Dashboard.md @@ -66,6 +66,14 @@ reloading while a permission prompt is blocking does not lose the red tab. Tabs can also be dragged to reorder. +### Automatic session names + +New sessions start with a short project/sequence name so they can be created immediately. +After the first task prompt is submitted, Codeman replaces that placeholder with a short +title derived locally from the prompt's first sentence. Slash commands such as `/clear` do +not become titles. A name you set with the inline rename action is treated as manual and is +never overwritten by automatic naming. + On phones the strip scrolls horizontally instead of wrapping, and the active tab is always scrolled into view. It is not reordered to the front, so the `Alt+N` numbering stays stable. diff --git a/src/session-auto-name.ts b/src/session-auto-name.ts new file mode 100644 index 00000000..82c57890 --- /dev/null +++ b/src/session-auto-name.ts @@ -0,0 +1,101 @@ +/** + * Helpers for assigning a useful default name after the first submitted prompt. + * + * This deliberately does not call an LLM: the prompt is already available at + * the input boundary, so a bounded local title is private, deterministic, and + * works for every CLI backend. + */ + +const MAX_PROMPT_BUFFER_LENGTH = 8_192; +const MAX_AUTO_NAME_CODE_POINTS = 72; + +/** + * Tracks terminal input until Enter is received. Terminal input arrives in + * arbitrary chunks, so this keeps only a small composer buffer and ignores + * navigation/control escape sequences. + */ +export class SubmittedPromptTracker { + private buffer = ''; + private escapeSequence = ''; + + feed(data: string): string[] { + const submitted: string[] = []; + + for (const character of data) { + if (this.escapeSequence) { + this.escapeSequence += character; + // CSI sequences end with a byte in the final-byte range. + const isCsiIntroducer = this.escapeSequence === '\x1b[' || this.escapeSequence === '\x1bO'; + if (/[\x40-\x7e]/.test(character) && !isCsiIntroducer) { + const isBracketedPasteMarker = this.escapeSequence === '\x1b[200~' || this.escapeSequence === '\x1b[201~'; + if (!isBracketedPasteMarker) this.buffer = ''; + this.escapeSequence = ''; + } + continue; + } + + if (character === '\x1b') { + this.escapeSequence = character; + continue; + } + + if (character === '\r' || character === '\n') { + const prompt = this.buffer.trim(); + if (prompt) submitted.push(prompt); + this.buffer = ''; + continue; + } + + if (character === '\x08' || character === '\x7f') { + this.buffer = Array.from(this.buffer).slice(0, -1).join(''); + continue; + } + + const codePoint = character.codePointAt(0) ?? 0; + if (codePoint < 0x20 || codePoint === 0x7f) { + // Ctrl-C/Ctrl-U and cursor controls make the append-only buffer + // unreliable. The next printable text starts a fresh candidate. + this.buffer = ''; + continue; + } + + this.buffer += character; + if (this.buffer.length > MAX_PROMPT_BUFFER_LENGTH) { + this.buffer = this.buffer.slice(-MAX_PROMPT_BUFFER_LENGTH); + } + } + + return submitted; + } +} + +/** + * Converts a submitted prompt into a compact session title. + * Returns null for empty text and slash commands, which are usually controls + * such as /clear or /resume rather than the task the user wants to remember. + */ +export function deriveAutoSessionName(prompt: string): string | null { + // Terminal input can legitimately contain ANSI/control bytes; they are + // removed before the title is persisted or broadcast. + const normalized = prompt + // eslint-disable-next-line no-control-regex + .replace(/\x1b\[[0-?]*[ -/]*[@-~]/g, '') + // eslint-disable-next-line no-control-regex + .replace(/[\u0000-\u001f\u007f]/g, ' ') + .replace(/\s+/g, ' ') + .trim(); + if (!normalized || normalized.startsWith('/')) return null; + + const firstSentence = normalized.match(/^.*?(?:[.!?。!?](?:\s|$)|$)/)?.[0]?.trim() || normalized; + const codePoints = Array.from(firstSentence); + if (codePoints.length <= MAX_AUTO_NAME_CODE_POINTS) return firstSentence; + return `${codePoints + .slice(0, MAX_AUTO_NAME_CODE_POINTS - 1) + .join('') + .trimEnd()}…`; +} + +/** Existing Codeman-generated tab names are safe to upgrade on first prompt. */ +export function isGeneratedSessionName(name: string): boolean { + return /^[ws]\d+-[a-zA-Z0-9_-]+$/.test(name); +} diff --git a/src/session.ts b/src/session.ts index 750690ea..68a493c6 100644 --- a/src/session.ts +++ b/src/session.ts @@ -23,7 +23,7 @@ * ralph-tracker (todo/completion parsing), bash-tool-parser (tool invocation tracking), * task-tracker (background tasks), mux-interface (tmux abstraction) * @consumedby session-manager, web/server, respawn-controller - * @emits session:terminal, session:idle, session:working, session:completion, session:exit + * @emits session:terminal, session:idle, session:working, session:completion, session:promptSubmitted, session:exit * * @module session */ @@ -56,6 +56,7 @@ import { type OmpConfig, type SessionRemote, type SessionDocker, + type SessionNameSource, } from './types.js'; import { resolveAndClaimOmpSessionId } from './utils/omp-session-resolver.js'; import { probeDockerCliVersion } from './docker-hosts.js'; @@ -116,6 +117,7 @@ import { SessionAutoOps } from './session-auto-ops.js'; import { detectUsageLimitPause } from './usage-limit-patterns.js'; import { SessionTaskCache } from './session-task-cache.js'; import { InteractivePtyExitBreaker } from './session-pty-exit-breaker.js'; +import { isGeneratedSessionName, SubmittedPromptTracker } from './session-auto-name.js'; import { parseTerminalAttachmentRequests } from './attachment-magic.js'; import { sanitizeAttachmentHistory, @@ -406,6 +408,8 @@ export class Session extends EventEmitter { private _taskCache = new SessionTaskCache(); private _name: string; + private _nameSource: SessionNameSource; + private readonly _submittedPromptTracker = new SubmittedPromptTracker(); private ptyProcess: pty.IPty | null = null; private _pid: number | null = null; private _status: SessionStatus = 'idle'; @@ -610,6 +614,8 @@ export class Session extends EventEmitter { workingDir: string; mode?: SessionMode; name?: string; + /** Whether the current name is still eligible for automatic replacement. */ + nameSource?: SessionNameSource; /** Terminal multiplexer instance (tmux) */ mux?: TerminalMultiplexer; /** Whether to use multiplexer wrapping */ @@ -677,6 +683,8 @@ export class Session extends EventEmitter { this.createdAt = config.createdAt || Date.now(); this.mode = config.mode || 'claude'; this._name = config.name || ''; + this._nameSource = + config.nameSource ?? (!this._name || isGeneratedSessionName(this._name) ? 'auto' : 'manual'); this._resumeSessionId = config.resumeSessionId; // NOW, not `createdAt`: recovery passes the ORIGINAL creation time of a // days-old tmux session, and seeding last-activity from it would report a @@ -1212,6 +1220,19 @@ export class Session extends EventEmitter { set name(value: string) { this._name = value; + this._nameSource = 'manual'; + } + + /** Replace an automatically generated name without taking ownership from auto naming. */ + applyAutoName(value: string): boolean { + const name = value.trim(); + if (!name || this._nameSource !== 'auto' || this._name === name) return false; + this._name = name; + return true; + } + + get nameSource(): SessionNameSource { + return this._nameSource; } setAutoClear(enabled: boolean, threshold?: number): void { @@ -1355,6 +1376,7 @@ export class Session extends EventEmitter { // attach repaint, so the home screens' quiet ordering survives a restart. lastActivityAt: this._wireActivityAt, name: this._name, + nameSource: this._nameSource, mode: this.mode, autoClearEnabled: this._autoOps.autoClearEnabled, autoClearThreshold: this._autoOps.autoClearThreshold, @@ -3204,9 +3226,10 @@ export class Session extends EventEmitter { * input could disappear while the caller believed it had been delivered. */ write(data: string): boolean { - this._trackSubmit(data); + const submittedPrompt = this._trackSubmit(data); if (!this.ptyProcess) return false; this.ptyProcess.write(data); + this._emitSubmittedPrompt(submittedPrompt); return true; } @@ -3223,10 +3246,18 @@ export class Session extends EventEmitter { return this._lastSubmitAt; } - private _trackSubmit(data: string): void { + private _trackSubmit(data: string): string[] { + const submitted = this._submittedPromptTracker.feed(data); if (data.includes('\r') || data.includes('\n')) { this._lastSubmitAt = Date.now(); } + return submitted; + } + + private _emitSubmittedPrompt(prompts: string[]): void { + for (const prompt of prompts) { + this.emit('promptSubmitted', prompt); + } } /** @@ -3299,13 +3330,16 @@ export class Session extends EventEmitter { * ``` */ async writeViaMux(data: string): Promise { - this._trackSubmit(data); + const submittedPrompt = this._trackSubmit(data); if (this._mux && this._muxSession) { - return this._mux.sendInput(this.id, data); + const sent = await this._mux.sendInput(this.id, data); + if (sent) this._emitSubmittedPrompt(submittedPrompt); + return sent; } // Fallback to PTY write if (this.ptyProcess) { this.ptyProcess.write(data); + this._emitSubmittedPrompt(submittedPrompt); return true; } return false; diff --git a/src/types/session.ts b/src/types/session.ts index 3c34b49d..d32ef0ce 100644 --- a/src/types/session.ts +++ b/src/types/session.ts @@ -58,6 +58,9 @@ export type SessionMode = | 'deepseek' | 'omp'; +/** Whether a session name may still be replaced by the first submitted prompt. */ +export type SessionNameSource = 'auto' | 'manual'; + export type RemoteCommandMode = Extract< SessionMode, 'shell' | 'claude' | 'opencode' | 'codex' | 'gemini' | 'antigravity' | 'pi' | 'grok' | 'deepseek' | 'omp' @@ -551,6 +554,8 @@ export interface SessionState { lastActivityAt: number; /** Session display name */ name?: string; + /** Name ownership; auto names are replaced after the first real prompt. */ + nameSource?: SessionNameSource; /** Session mode */ mode?: SessionMode; /** Auto-clear enabled */ diff --git a/src/web/routes/session-routes.ts b/src/web/routes/session-routes.ts index 84584990..6b8d0aa7 100644 --- a/src/web/routes/session-routes.ts +++ b/src/web/routes/session-routes.ts @@ -1103,6 +1103,7 @@ export function registerSessionRoutes( workingDir, mode, name: body.name || '', + nameSource: body.name ? undefined : 'auto', mux: ctx.mux, useMux: true, niceConfig: globalNice, @@ -3333,6 +3334,7 @@ export function registerSessionRoutes( const session = new Session({ workingDir: resolvedCasePath, name: sessionName ? sessionName.slice(0, MAX_SESSION_NAME_LENGTH) : '', + nameSource: sessionName ? undefined : 'auto', mux: ctx.mux, useMux: true, mode: mode, diff --git a/src/web/server.ts b/src/web/server.ts index 0cf318d6..08794f1c 100644 --- a/src/web/server.ts +++ b/src/web/server.ts @@ -1620,6 +1620,7 @@ export class WebServer extends EventEmitter { getStore: () => this.store, registerAttachment: (id: string, filePath: string, source: 'external' | 'codex-generated') => this.registerAttachment(id, filePath, source), + updateSessionName: (id: string, name: string) => this.mux.updateSessionName(id, name), }; } @@ -2768,6 +2769,7 @@ export class WebServer extends EventEmitter { workingDir: muxSession.workingDir, mode: muxSession.mode, name: sessionName, + nameSource: savedState?.nameSource, // When the session FIRST started, not when this server booted. // Without it every recovered session was restamped `Date.now()` on // each restart, so a week-old pane read as "created 2m ago" on the diff --git a/src/web/session-listener-wiring.ts b/src/web/session-listener-wiring.ts index 33724557..2af8d7a0 100644 --- a/src/web/session-listener-wiring.ts +++ b/src/web/session-listener-wiring.ts @@ -3,7 +3,7 @@ * * Extracted from server.ts for modularity. Provides: * - `SessionListenerRefs` interface (named listener references for leak-free cleanup) - * - `createSessionListeners()` — builds all 25 listener handlers via dependency injection + * - `createSessionListeners()` — builds all session listener handlers via dependency injection * - `attachSessionListeners()` / `detachSessionListeners()` — symmetric attach/detach * * The detach function deduplicates a pattern that was previously copy-pasted 3 times @@ -29,6 +29,7 @@ import { getLifecycleLog } from '../session-lifecycle-log.js'; import { fileStreamManager } from '../file-stream-manager.js'; import { sessionWaits } from './session-wait-registry.js'; import { approvalInbox } from './approval-inbox.js'; +import { deriveAutoSessionName } from '../session-auto-name.js'; /** Stored listener references for session cleanup (prevents memory leaks) */ export interface SessionListenerRefs { @@ -63,6 +64,7 @@ export interface SessionListenerRefs { bashToolEnd: (tool: ActiveBashTool) => void; bashToolsUpdate: (tools: ActiveBashTool[]) => void; attachmentRequested: (event: { path: string; source: 'external' | 'codex-generated' }) => void; + promptSubmitted: (prompt: string) => void; } /** Dependencies injected by WebServer — keeps listener creation decoupled from server internals. */ @@ -83,10 +85,11 @@ interface SessionListenerDeps { cleanupRespawnOnExit(sessionId: string): void; getStore(): import('../state-store.js').StateStore; registerAttachment(sessionId: string, filePath: string, source: 'external' | 'codex-generated'): Promise; + updateSessionName(sessionId: string, name: string): boolean; } /** - * Creates all 26 session listener handlers, capturing dependencies via closure. + * Creates all session listener handlers, capturing dependencies via closure. * Call `attachSessionListeners()` after to wire them to the session. */ export function createSessionListeners(session: Session, deps: SessionListenerDeps): SessionListenerRefs { @@ -451,6 +454,15 @@ export function createSessionListeners(session: Session, deps: SessionListenerDe console.error(`[Attachment] Failed to register ${event.path} for ${session.id}:`, err); }); }, + + /** Assigns a bounded local title from the first real task prompt. */ + promptSubmitted: (prompt: string) => { + const name = deriveAutoSessionName(prompt); + if (!name || !session.applyAutoName(name)) return; + deps.updateSessionName(session.id, session.name); + deps.persistSessionState(session); + deps.broadcast(SseEvent.SessionUpdated, deps.getSessionStateWithRespawn(session)); + }, }; } @@ -487,6 +499,7 @@ export function attachSessionListeners(session: Session, refs: SessionListenerRe session.on('bashToolEnd', refs.bashToolEnd); session.on('bashToolsUpdate', refs.bashToolsUpdate); session.on('attachmentRequested', refs.attachmentRequested); + session.on('promptSubmitted', refs.promptSubmitted); } /** Detach all listeners from a session (prevents memory leaks from closure references). */ @@ -522,4 +535,5 @@ export function detachSessionListeners(session: Session, refs: SessionListenerRe session.off('bashToolEnd', refs.bashToolEnd); session.off('bashToolsUpdate', refs.bashToolsUpdate); session.off('attachmentRequested', refs.attachmentRequested); + session.off('promptSubmitted', refs.promptSubmitted); } diff --git a/test/session-listener-wiring.test.ts b/test/session-listener-wiring.test.ts index 74ea312c..5ffd56e0 100644 --- a/test/session-listener-wiring.test.ts +++ b/test/session-listener-wiring.test.ts @@ -20,4 +20,31 @@ describe('session listener wiring', () => { ); expect(registerAttachment).toHaveBeenNthCalledWith(2, 'wiring-attach-source-test', '/tmp/report.pdf', 'external'); }); + + it('renames an eligible session when its first prompt is submitted', () => { + const session = new Session({ id: 'wiring-auto-name-test', workingDir: '/tmp', name: 'w1-demo' }); + const updateSessionName = vi.fn(() => true); + const persistSessionState = vi.fn(); + const broadcast = vi.fn(); + const getSessionStateWithRespawn = vi.fn(() => session.toState()); + const deps = { + updateSessionName, + persistSessionState, + broadcast, + getSessionStateWithRespawn, + } as unknown as Parameters[1]; + + const refs = createSessionListeners(session, deps); + refs.promptSubmitted('整理登录模块并补充测试'); + + expect(session.name).toBe('整理登录模块并补充测试'); + expect(updateSessionName).toHaveBeenCalledWith('wiring-auto-name-test', '整理登录模块并补充测试'); + expect(persistSessionState).toHaveBeenCalledWith(session); + expect(broadcast).toHaveBeenCalled(); + + session.name = '人工命名'; + refs.promptSubmitted('新的任务不能覆盖人工命名'); + expect(session.name).toBe('人工命名'); + expect(updateSessionName).toHaveBeenCalledTimes(1); + }); }); diff --git a/test/session-submit-anchor.test.ts b/test/session-submit-anchor.test.ts index a1298fb9..5c75564c 100644 --- a/test/session-submit-anchor.test.ts +++ b/test/session-submit-anchor.test.ts @@ -12,6 +12,7 @@ import { describe, it, expect } from 'vitest'; import { Session } from '../src/session.js'; +import { deriveAutoSessionName, SubmittedPromptTracker } from '../src/session-auto-name.js'; describe('session submit anchor', () => { it('records the pane Enter and carries it into persisted state', () => { @@ -53,3 +54,31 @@ describe('session submit anchor', () => { expect(recovered.lastSubmitAt).toBe(0); }); }); + +describe('automatic session names', () => { + it('builds a bounded title from the first sentence without exposing controls', () => { + expect(deriveAutoSessionName(' 修复登录跳转问题。\n不要改数据库')).toBe('修复登录跳转问题。'); + expect(deriveAutoSessionName('/clear')).toBeNull(); + expect(deriveAutoSessionName('\x1b[31m整理项目文档\x1b[0m')).toBe('整理项目文档'); + expect(Array.from(deriveAutoSessionName('a'.repeat(200)) ?? '')).toHaveLength(72); + }); + + it('tracks chunked typing, backspace, and Enter without treating arrows as prompt text', () => { + const tracker = new SubmittedPromptTracker(); + expect(tracker.feed('修复登')).toEqual([]); + expect(tracker.feed('录跳转\x7f问题\r')).toEqual(['修复登录跳问题']); + expect(tracker.feed('旧内容\x1b[A新内容\r')).toEqual(['新内容']); + }); + + it('keeps manual names protected while generated names remain eligible', () => { + const generated = new Session({ workingDir: '/tmp', name: 'w1-demo' }); + expect(generated.nameSource).toBe('auto'); + expect(generated.applyAutoName('修复登录')).toBe(true); + expect(generated.applyAutoName('继续重命名')).toBe(true); + + const manual = new Session({ workingDir: '/tmp', name: '我的工作窗口' }); + expect(manual.nameSource).toBe('manual'); + expect(manual.applyAutoName('不应覆盖')).toBe(false); + expect(manual.name).toBe('我的工作窗口'); + }); +}); From de864e7d63e499d83dc924050109864bd3ae537f Mon Sep 17 00:00:00 2001 From: Codeman maintainer Date: Mon, 14 Sep 2026 23:40:10 +0200 Subject: [PATCH 02/28] fix(terminal): restore the history anchor after xterm parses, not before flushPendingWrites() captured the viewport of a user who was reading scrollback, called terminal.write(), and restored the anchor on the next line. xterm parses on its own schedule, so at that point the buffer has not moved: the guard `viewportY !== preserveViewportY` was false, scrollToLine was never called at all, and the Codex redraw landed a tick later and took the viewport to the live bottom with nothing left to pull it back. Scrolling up during a stream still got dragged down, which is what #358 reports, and a refresh was the only way back to a coherent view. The restore moves inside xterm's write callback, the first moment the redraw's effect exists, and runs before _scheduleTerminalWriteFlush() so a deferred remainder re-captures the restored anchor rather than the bottom. Two things follow from it running later: - A live anchor now wins over the sticky scroll-to-bottom. The two are captured at different moments (_wasAtBottomBeforeWrite at the frame's first batchTerminalWrite, the anchor at flush time), so a scroll-up in between leaves both set, and running both would jump to the bottom and come back a frame later instead of staying put. - The anchor is dropped if the active session changed or a buffer load started while the write was in flight. It indexes the buffer it was captured from, and selectSession() resets the terminal and chunk-loads a different scrollback. The existing regression passed throughout, because its write mock moved the viewport synchronously, which real xterm never does. The harness now models an asynchronous parse (redraw lands, then the callback fires), and all five of the anchor tests fail against the old code. Fixes #358 Co-Authored-By: Claude Opus 5 (1M context) --- .../terminal-history-anchor-after-parse.md | 5 + src/web/public/terminal-ui.js | 50 ++++++- test/terminal-flush-budget.test.ts | 140 +++++++++++++++++- 3 files changed, 179 insertions(+), 16 deletions(-) create mode 100644 .changeset/terminal-history-anchor-after-parse.md diff --git a/.changeset/terminal-history-anchor-after-parse.md b/.changeset/terminal-history-anchor-after-parse.md new file mode 100644 index 00000000..418171c4 --- /dev/null +++ b/.changeset/terminal-history-anchor-after-parse.md @@ -0,0 +1,5 @@ +--- +"aicodeman": patch +--- + +Keep the terminal anchored where you are reading while an agent streams (#358). Scrolling up during a Codex response could still be dragged back to the live bottom by the next redraw: the flush captured the viewport before writing and restored it immediately after, but xterm parses asynchronously, so at that moment the buffer had not moved yet, the restore compared the anchor against itself and did nothing, and the redraw landed a tick later with nothing left to pull the view back. The restore now runs inside xterm's own write callback, which is the first point at which the redraw's effect exists, and it holds across consecutive and chunked redraws. It is dropped if you switch sessions or a history replay starts before the write parses, since the anchor indexes the buffer it was captured from. diff --git a/src/web/public/terminal-ui.js b/src/web/public/terminal-ui.js index 7ecc994c..e88e1c07 100644 --- a/src/web/public/terminal-ui.js +++ b/src/web/public/terminal-ui.js @@ -3543,6 +3543,29 @@ Object.assign(CodemanApp.prototype, { this._sendInputAsync(this.activeSessionId, text); }, + /** + * Re-assert a history anchor captured before a terminal write (#358). + * + * Called from xterm's write callback, never synchronously after write(): + * xterm parses on its own schedule, so the buffer only carries the redraw's + * effect once that callback fires. A null anchor means the user was following + * live output and nothing needs restoring. + */ + _restoreTerminalViewport(preserveViewportY, sessionId) { + if (preserveViewportY === null || preserveViewportY === undefined) return; + // The anchor is a row index into the buffer it was captured from. Now that + // this runs a parse later instead of synchronously, a session switch can land + // in between: selectSession() resets the terminal and chunk-loads the new + // session's scrollback, and scrolling THAT buffer to a row that meant + // something in the previous one is not a restore, it is a jump to an + // arbitrary place. Both checks cover one half of that window. + if (sessionId !== undefined && sessionId !== this.activeSessionId) return; + if (this._isLoadingBuffer) return; + if (typeof this.terminal?.scrollToLine !== 'function') return; + if (this.terminal.buffer?.active?.viewportY === preserveViewportY) return; + this.terminal.scrollToLine(preserveViewportY); + }, + /** * Flush pending writes to terminal, processing DEC 2026 sync markers. * Strips markers and writes content atomically within a single frame. @@ -3578,6 +3601,8 @@ Object.assign(CodemanApp.prototype, { // scroll-to-bottom below, where it protects against a mid-flush race. const preserveViewportY = this.terminal.buffer?.active && !this.isTerminalAtBottom() ? this.terminal.buffer.active.viewportY : null; + // Which buffer the anchor belongs to, checked again when the write parses. + const flushSessionId = this.activeSessionId; const writeChunk = joined.slice(0, MAX_FRAME_BYTES); if (_joinedLen > MAX_FRAME_BYTES) { @@ -3592,6 +3617,16 @@ Object.assign(CodemanApp.prototype, { this.terminal.write(writeChunk, () => { this._terminalWriteInFlight = false; this._terminalWriteInFlightBytes = 0; + // Restore INSIDE the callback (#358). xterm parses asynchronously, so + // the moment write() returns the buffer has not moved yet: the old + // restore ran here, found viewportY still equal to the anchor, and did + // nothing at all — then the parse landed and a cursor-addressed Codex + // redraw dragged the viewport to the live bottom with nothing left to + // pull it back. The callback is xterm's own "this chunk is parsed" + // signal, which is the earliest point the anchor can actually be + // reasserted. (The synchronous version passed its regression test only + // because the test's write mock moved the viewport synchronously.) + this._restoreTerminalViewport(preserveViewportY, flushSessionId); this._scheduleTerminalWriteFlush(); }); } catch (err) { @@ -3599,13 +3634,6 @@ Object.assign(CodemanApp.prototype, { this._terminalWriteInFlightBytes = 0; throw err; } - if ( - preserveViewportY !== null && - this.terminal.buffer?.active?.viewportY !== preserveViewportY && - typeof this.terminal.scrollToLine === 'function' - ) { - this.terminal.scrollToLine(preserveViewportY); - } const bytesThisFrame = deferred ? MAX_FRAME_BYTES : _joinedLen; const _dt = performance.now() - _t0; if (_dt > 100 || deferred) @@ -3617,7 +3645,13 @@ Object.assign(CodemanApp.prototype, { // Give manual scroll-up gestures a short grace window so high-frequency // Codex status ticks do not snap the viewport back while the user is // trying to inspect earlier output. - if (this._wasAtBottomBeforeWrite && !this._hasRecentUserScrollUp()) { + // + // A live anchor wins outright. The two flags are captured at different + // moments (_wasAtBottomBeforeWrite at the frame's first batchTerminalWrite, + // the anchor at flush time), so a scroll-up in between leaves both set; now + // that the anchor is reasserted after the parse, running both would jump to + // the bottom and then back one frame later instead of simply staying put. + if (preserveViewportY === null && this._wasAtBottomBeforeWrite && !this._hasRecentUserScrollUp()) { this.terminal.scrollToBottom(); } diff --git a/test/terminal-flush-budget.test.ts b/test/terminal-flush-budget.test.ts index 514e93da..82e0575c 100644 --- a/test/terminal-flush-budget.test.ts +++ b/test/terminal-flush-budget.test.ts @@ -49,6 +49,39 @@ function loadTerminalUiHarness(mode: string) { return { app, writes }; } +/** + * Swap in a terminal whose write() parses ASYNCHRONOUSLY, the way xterm.js does. + * + * The real renderer queues the chunk and applies it later, firing the write + * callback once it has been parsed; a redraw that addresses a row past the + * viewport (Codex's status line) drags the viewport to the live bottom at that + * point, not when write() returns. `parse()` runs that pending work. + */ +function attachAsyncParsingTerminal(app: any, opts: { viewportY: number; baseY: number }) { + const buffer = { viewportY: opts.viewportY, baseY: opts.baseY }; + const pending: Array<() => void> = []; + app.terminal.buffer = { active: buffer }; + app.terminal.write = vi.fn((_data: string, callback?: () => void) => { + pending.push(() => { + buffer.viewportY = buffer.baseY; // the redraw lands + callback?.(); + }); + }); + app.terminal.scrollToLine = vi.fn((line: number) => { + buffer.viewportY = line; + }); + app.terminal.scrollToBottom = vi.fn(() => { + buffer.viewportY = buffer.baseY; + }); + return { + buffer, + parse: () => { + const queued = pending.splice(0, pending.length); + for (const run of queued) run(); + }, + }; +} + function loadAppHarness() { const dir = resolve(import.meta.dirname, '../src/web/public'); const fetchMock = vi.fn(); @@ -337,20 +370,111 @@ describe('terminal flush budget', () => { it('restores the user scroll position when Codex Working redraws move the viewport', () => { const { app } = loadTerminalUiHarness('codex'); - const buffer = { viewportY: 40, baseY: 100 }; - app.terminal.buffer = { active: buffer }; - app.terminal.write = vi.fn(() => { - buffer.viewportY = buffer.baseY; - }); - app.terminal.scrollToLine = vi.fn((line: number) => { - buffer.viewportY = line; - }); + const { buffer, parse } = attachAsyncParsingTerminal(app, { viewportY: 40, baseY: 100 }); app._wasAtBottomBeforeWrite = true; app._lastUserScrollUpAt = 0; app.pendingWrites.push('\x1b[55;1H\x1b[2m• Working (6s)'); app.flushPendingWrites(); + parse(); expect(buffer.viewportY).toBe(40); }); + + // Issue #358. xterm.js parses on its own schedule, so the buffer still holds + // the pre-write viewport the instant write() returns: restoring there compared + // the anchor against itself, did nothing, and left the redraw free to drag the + // viewport to the live bottom a tick later. The previous regression passed + // because its write mock moved the viewport synchronously, which real xterm + // never does. These drive the callback explicitly instead. + it('restores the history anchor only AFTER xterm has parsed the write (#358)', () => { + const { app } = loadTerminalUiHarness('codex'); + const { buffer, parse } = attachAsyncParsingTerminal(app, { viewportY: 40, baseY: 100 }); + app.pendingWrites.push('\x1b[55;1H\x1b[2m• Working (6s)'); + + app.flushPendingWrites(); + // Nothing has parsed yet, so nothing may have been restored yet either. + expect(app.terminal.scrollToLine).not.toHaveBeenCalled(); + expect(buffer.viewportY).toBe(40); + + parse(); + + expect(app.terminal.scrollToLine).toHaveBeenCalledWith(40); + expect(buffer.viewportY).toBe(40); + }); + + it('holds the anchor across consecutive Codex redraws', () => { + const { app } = loadTerminalUiHarness('codex'); + const { buffer, parse } = attachAsyncParsingTerminal(app, { viewportY: 40, baseY: 100 }); + + for (const frame of ['\x1b[55;1H\x1b[2m• Working (6s)', '\x1b[55;1H\x1b[2m• Working (7s)']) { + app.pendingWrites.push(frame); + app.flushPendingWrites(); + parse(); + expect(buffer.viewportY).toBe(40); + } + }); + + it('holds the anchor across a chunked write whose remainder is deferred', () => { + const { app } = loadTerminalUiHarness('codex'); + const { buffer, parse } = attachAsyncParsingTerminal(app, { viewportY: 40, baseY: 100 }); + // Over the 32KB codex frame budget, so the flush defers a remainder and the + // second chunk goes out from the write callback's reschedule. + app.pendingWrites.push('x'.repeat(40000)); + + app.flushPendingWrites(); + parse(); + expect(buffer.viewportY).toBe(40); + + app.flushPendingWrites(); + parse(); + expect(buffer.viewportY).toBe(40); + expect(app.pendingWrites).toHaveLength(0); + }); + + it('drops the anchor when the user switched sessions before the write parsed', () => { + // The anchor indexes the buffer it came from. selectSession() resets the + // terminal and chunk-loads a different scrollback, so replaying row 40 into + // that one is a jump to an arbitrary place, not a restore. Only reachable now + // that the restore runs a parse later than the write. + const { app } = loadTerminalUiHarness('codex'); + const { buffer, parse } = attachAsyncParsingTerminal(app, { viewportY: 40, baseY: 100 }); + app.pendingWrites.push('\x1b[55;1H\x1b[2m• Working (6s)'); + + app.flushPendingWrites(); + app.activeSessionId = 'session-2'; // the user clicked another tab + parse(); + + expect(app.terminal.scrollToLine).not.toHaveBeenCalled(); + expect(buffer.viewportY).toBe(buffer.baseY); + }); + + it('drops the anchor while a buffer load is replaying history', () => { + const { app } = loadTerminalUiHarness('codex'); + const { parse } = attachAsyncParsingTerminal(app, { viewportY: 40, baseY: 100 }); + app.pendingWrites.push('\x1b[55;1H\x1b[2m• Working (6s)'); + + app.flushPendingWrites(); + app._isLoadingBuffer = true; // chunkedTerminalWrite owns the viewport now + parse(); + + expect(app.terminal.scrollToLine).not.toHaveBeenCalled(); + }); + + it('does not bounce off the bottom when the sticky flag and an anchor disagree', () => { + // _wasAtBottomBeforeWrite is captured at the frame's first batchTerminalWrite + // and the anchor at flush time, so a scroll-up in between leaves both live. + // The anchor wins: scrolling to the bottom and back would be a visible jump. + const { app } = loadTerminalUiHarness('codex'); + const { buffer, parse } = attachAsyncParsingTerminal(app, { viewportY: 40, baseY: 100 }); + app._wasAtBottomBeforeWrite = true; + app._lastUserScrollUpAt = 0; + app.pendingWrites.push('\x1b[55;1H\x1b[2m• Working (6s)'); + + app.flushPendingWrites(); + parse(); + + expect(app.terminal.scrollToBottom).not.toHaveBeenCalled(); + expect(buffer.viewportY).toBe(40); + }); }); From 018f0c4160279414bca4b546049e069efc6325ef Mon Sep 17 00:00:00 2001 From: Codeman maintainer Date: Tue, 15 Sep 2026 00:56:15 +0200 Subject: [PATCH 03/28] docs(readme): catch both READMEs up to 1.29.0 and repair three merge-damaged lines DeepSeek Harness joins every CLI list it was missing from (tagline, intro, run-mode table, Multi-CLI bullet with its env prefixes, security allowlist, architecture diagram), and the 1.27 to 1.29.0 features get their bullets: custom model endpoints (HTTP API only, with the verified and gapped CLIs named), web tabs, attaching a case to an existing container, remote SSH file access, the plan-usage chip, the sidebar and activity-sorted rail, font weight, skins and entrance animations, Approvals Inbox, Read My Mind, Claude-login voice dictation, Shift+drag select and right-click copy. The agent guide's rule 7 now counts deepseek among the hook-signalling modes and the recipes read answers through last-response first; the API section carries the new routes and current counts; the download cap reads 2 GB instead of the retired 50 MB; the zerolag package test count is the measured 238. The English file had three spots where the OMP merge of 2026-08-18 left two copies of a line joined without a newline (the Docker credentials bullet, rule 7 of the agent guide, the CLI node of the mermaid diagram). All three are single lines again. The Chinese file was further behind: besides the above it had never received the daemon and service block, the Tailscale install option, the Compose paragraph, the Tab Alerts section, the codeman tui section, the agent-skill walkthrough, the Community section or the closing star paragraph. Those are translated in, so both files now share one section structure. Co-Authored-By: Claude Fable 5.1 --- README.md | 87 ++++++++++------ README.zh-CN.md | 258 ++++++++++++++++++++++++++++++++++++++---------- 2 files changed, 264 insertions(+), 81 deletions(-) diff --git a/README.md b/README.md index 27ba7db2..31acc908 100644 --- a/README.md +++ b/README.md @@ -5,7 +5,7 @@

Mission control for AI coding agents

- Claude Code • OpenCode • Codex • Antigravity • Gemini • Pi • Grok • OMP • Terminal - One Dashboard • Any Device + Claude Code • OpenCode • Codex • Antigravity • Gemini • Pi • Grok • DeepSeek • OMP • Terminal - One Dashboard • Any Device

@@ -27,7 +27,7 @@ Codeman — parallel subagent visualization

-**Codeman** is a self-hosted mission control for AI coding agents. It spawns Claude Code, OpenCode, Codex, Antigravity, Gemini, Pi, Grok, or OMP inside persistent tmux sessions, streams the real terminal to any browser, and keeps agents productive after you walk away: it re-prompts on idle, resumes when a usage limit resets, runs scheduled jobs, and shows every background agent working in real time. +**Codeman** is a self-hosted mission control for AI coding agents. It spawns Claude Code, OpenCode, Codex, Antigravity, Gemini, Pi, Grok, DeepSeek Harness, or OMP inside persistent tmux sessions, streams the real terminal to any browser, and keeps agents productive after you walk away: it re-prompts on idle, resumes when a usage limit resets, runs scheduled jobs, and shows every background agent working in real time. Get started in one line (macOS & Linux, Windows via WSL): @@ -42,7 +42,7 @@ codeman web The installer asks before every system change, and re-running the same line updates in place. Full details: [Quick Start - Installation](#quick-start---installation). -- **One dashboard, eight CLIs** - run [Claude Code, OpenCode, Codex, Antigravity, Gemini, Pi, Grok, or OMP](#more-features) per session (plus plain shell), locally, [in Docker](#isolated-docker-sessions), or [over SSH](#remote-ssh-sessions) +- **One dashboard, nine CLIs** - run [Claude Code, OpenCode, Codex, Antigravity, Gemini, Pi, Grok, DeepSeek, or OMP](#more-features) per session (plus plain shell), locally, [in Docker](#isolated-docker-sessions), or [over SSH](#remote-ssh-sessions), with your own dashboards open as [web tabs](#more-features) beside them - **Truly phone-friendly** - a [touch-optimized terminal](#mobile-optimized-web-ui) with instant local echo, QR login, swipe navigation, and push notifications - **Runs while you sleep** - [idle detection + respawn cycling](#respawn-controller) and auto-resume when a subscription limit resets, for 24+ hour unattended runs - **See your agents think** - [live floating windows](#live-agent-visualization) for every subagent and teammate, with real-time transcripts @@ -68,7 +68,7 @@ This installs Node.js, tmux and a build toolchain if missing (node-pty ships no - **Re-run to update.** The same one-liner updates a finished install in place: local changes in `~/.codeman/app` are stashed (never discarded), and a running service is restarted and verified. If a first install was interrupted, re-running resumes the full setup instead. `install.sh update` and `install.sh uninstall` also exist. - **CI / headless:** without a terminal attached, steps that would change your system abort with instructions instead of running silently. Set `CODEMAN_NONINTERACTIVE=1` to approve them for automation. -You'll need at least one AI coding CLI installed — [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), [Gemini CLI](https://github.com/google-gemini/gemini-cli), [Pi](https://pi.dev), [Grok Build](https://github.com/xai-org/grok-build), [DeepSeek Harness](https://github.com/deepseek-ai/deepseek-harness), or [OMP](https://github.com/can1357/oh-my-pi) (any combination works; Gemini CLI is enterprise-only since Google's consumer cutover, and Antigravity is its successor). The installer detects whichever of the nine is present; if none is found, it offers to install Claude Code or OpenCode, or you can skip and install one yourself later. After install: +You'll need at least one AI coding CLI installed — [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), [Gemini CLI](https://github.com/google-gemini/gemini-cli), [Pi](https://pi.dev), [Grok Build](https://github.com/xai-org/grok-build), [DeepSeek Harness](https://github.com/deepseek-ai/deepseek-harness), or [OMP](https://github.com/can1357/oh-my-pi) (any combination works; Gemini CLI is enterprise-only since Google's consumer cutover, and Antigravity is its successor). The installer detects whichever of the nine is present; if none is found, it offers to install any of them from a menu (DeepSeek excepted, since its npm package installs only a launcher with no runnable profile), or you can skip and install one yourself later. After install: ```bash codeman web @@ -82,7 +82,7 @@ codeman users add alice --admin # create the first admin account codeman web --multiuser # named logins + per-user case spaces ``` -**Prefer Docker Compose?** A local-image Compose deployment ships in `docker/`: copy `docker/.env.example` to `docker/.env`, set `CODEMAN_PASSWORD`, then run `bash docker/Start-Codeman.sh` on Linux. Codeman runs in a container and spawns Docker cases as sibling containers through the host socket. See the [Docker deployment guide](docker/README.md) for direct Compose commands, storage and networking options. +**Prefer Docker Compose?** A local-image Compose deployment ships in `docker/`: copy `docker/.env.example` to `docker/.env`, set `CODEMAN_PASSWORD`, then run `bash docker/Start-Codeman.sh` on Linux. Codeman runs in a container and spawns Docker cases as sibling containers through the host socket. After updating, run the script again rather than a plain `docker compose up`, so the rebuilt image, refreshed volumes and entrypoint arrive together. See the [Docker deployment guide](docker/README.md) for direct Compose commands, storage and networking options. Details in [Multi-User Mode](#multi-user-mode-opt-in) below. @@ -212,7 +212,7 @@ The most responsive AI coding agent experience on any phone. Full xterm.js termi - **Keyboard accessory bar** — `/init`, `/clear`, `/compact` quick-action buttons above the virtual keyboard; destructive commands require a double-press to confirm, so you never fire one by accident; on Codex sessions the bar also shows `⇧←` / `⇧→` (Shift+Left / Shift+Right: edit the last queued message / return through the prompt stack) - **Dedicated Enter button** — replays the keypress through the terminal, so text buffered by local echo is flushed first rather than stranded - **Swipe navigation & smart keyboard handling** — swipe left/right to switch sessions; toolbar and terminal shift up when the keyboard opens (`visualViewport` API) -- **Built for phones** — safe-area insets for notch and home indicator, 44px touch targets, bottom-sheet case picker, native momentum scrolling +- **Built for phones** — safe-area insets for notch and home indicator, 44px touch targets, bottom-sheet case picker, native momentum scrolling; on a folding phone (iPhone Duo) dialogs stay clear of the hinge, and opening or closing the device is never mistaken for the keyboard ```bash codeman web --https @@ -255,7 +255,7 @@ Click **+ New Session** (or **Quick Start**). A session is one AI CLI running in | Field | What it does | | ---------------------------- | ------------------------------------------------------------------------------------------------------------------- | | **Working directory / case** | The folder the agent operates in. A "case" is just a named working dir Codeman remembers. **Add Case** creates one from scratch, links an existing folder, or clones a GitHub repo straight into one (**Clone Repo**). | -| **CLI / run mode** | `Claude` (default), `OpenCode`, `Codex`, `Antigravity`, `Gemini`, `Pi`, `Grok`, `OMP`, or `Terminal` (plain shell). | +| **CLI / run mode** | `Claude` (default), `OpenCode`, `Codex`, `Antigravity`, `Gemini`, `Pi`, `Grok`, `DeepSeek`, `OMP`, or `Terminal` (plain shell). | | **Model** | Per-session model (App Settings → Models → New Claude sessions). A soft default — `/model` still works in-session. | | **Effort / Ultracode** | Reasoning effort (`low`–`max`) or `ultracode` for dynamic multi-agent workflows. Switchable anytime with `/effort`. | @@ -263,7 +263,7 @@ Hit start — Codeman spawns the CLI via a real PTY and streams it to your brows ### 3. Read the dashboard -- **Tabs (top)** — one per session. `Alt+1`-`9` to jump, `Ctrl+Tab` for next, drag to reorder (tab order syncs across your devices). +- **Tabs (top)** — one per session. `Alt+1`-`9` to jump, `Ctrl+Tab` for next, drag to reorder (tab order syncs across your devices). Prefer a list? **App Settings → Appearance → Tabs** moves it into a left sidebar with a filter box (`Alt+B` collapses it) or a vertical rail whose rows sort by activity: blocked on you first, then longest running, then most recently quiet. - **Terminal (center)** — a real `xterm.js` terminal; full TUIs render correctly. Type directly and press **Enter** to send. `Shift+Enter` inserts a newline. - **Side panels** — Respawn, Orchestrator, Cron, Subagents, Settings (toggled from the toolbar). @@ -271,8 +271,10 @@ Hit start — Codeman spawns the CLI via a real PTY and streams it to your brows - **Type prompts** straight into the terminal — input is delivered exactly-once even across reconnects (a dropped link never loses or double-sends a prompt). - **Paste or drag-and-drop images** directly into the session. -- **Voice input** — `Ctrl+Shift+V` (Deepgram Nova-3, with auto-silence stop). -- **Attachments** — register external files/docs and preview Office/PDF inline. +- **Voice input** — `Ctrl+Shift+V` (Deepgram Nova-3, or this machine's Claude Code login with no API key; auto-silence stop). +- **Attachments** — register external files/docs and preview Office/PDF inline; any file path an agent prints is clickable, in the terminal and in the chat view. +- **When it needs you** — the tab turns yellow (waiting for input) or red (a question is blocking). The **Approvals Inbox** _(opt-in)_ queues every pending prompt across sessions, answerable from the header bell or the phone home screen, and 🧠 **Read My Mind** _(opt-in)_ drafts your next prompt from the case's goals and recent work. +- **Copy what you see** — `Shift+drag` selects text even while the CLI owns the mouse, right-click copies it, and Auto Copy _(opt-in)_ copies a selection the moment you release it. ### 5. Make it autonomous @@ -291,7 +293,7 @@ Hit start — Codeman spawns the CLI via a real PTY and streams it to your brows ### 7. Operate & maintain -- **App Settings** — model, effort, permission startup mode, theme/skin, notifications, display toggles, per-CLI options, a synced custom display name, and per-device English/Simplified Chinese UI language. +- **App Settings** — model, effort, permission startup mode, theme/skin, terminal font family and weight, entrance animations, notifications, display toggles, per-CLI options, a synced custom display name, and per-device English/Simplified Chinese UI language. - **Run it in the background** — `codeman web -d` detaches from your shell (`--status`, `--stop`); `codeman service install` makes it a systemd user unit / macOS LaunchAgent that survives reboots. Both verify the server actually answers before reporting success, and both refuse to start a second server on one data dir. See [Keep it running in the background](#quick-start---installation). - **Self-update** — git-clone installs update in place from **App Settings → System → Updates**. - **Deploy your own changes** — see [Development](#development). @@ -439,16 +441,21 @@ PTY Output → 16ms Server Batch → DEC 2026 Wrap → SSE → Client rAF → xt - **Background daemon & service install** — `codeman web -d` runs the server detached with a pidfile, `~/.codeman/web.log`, and verified startup (it polls the server until it answers, so a port clash never reads as success); `codeman service install` writes a systemd user unit (Linux) or LaunchAgent (macOS) with your shell's PATH baked in, so an nvm or Homebrew `node`, `tmux` and `claude` are actually found. Secrets are never written into unit files - **Self-update** — git-clone installs under systemd/launchd update in place from **App Settings → System → Updates**: it detects the latest release, auto-stashes a dirty tree, and streams build progress across the service restart (npm installs report as non-updatable) - **Clone a GitHub repo as a case** — paste a repository URL into **Add Case → Clone Repo** and Codeman clones it into `~/codeman-cases/` and registers it as a normal case, ready to run an agent in. It preflights the URL while you type (tells you whether it can be cloned anonymously and offers the repo's real branches and tags for the optional branch/tag field), fills the case name in from the URL, and lets you pick which CLI the Run button should use. Public repositories over `https://`; Codeman never collects or stores credentials -- **Multi-CLI** — run **Claude Code**, **OpenCode**, **Codex**, **Antigravity**, **Gemini**, **Pi**, **Grok**, or **OMP** per session; env-var prefixes auto-gate (`CLAUDE_CODE_*` vs `OPENCODE_*` vs `CODEX_*` vs `ANTIGRAVITY_*` vs `GEMINI_*`/`GOOGLE_*` vs `PI_*` vs `GROK_*`/`XAI_*` vs `OMP_*`). See [`docs/opencode-integration.md`](docs/opencode-integration.md), [`docs/pi-integration.md`](docs/pi-integration.md), [`docs/grok-integration.md`](docs/grok-integration.md) and [`docs/omp-integration.md`](docs/omp-integration.md) -- **Docker sessions** — run a case inside an isolated, hardened container. One checkbox on **Create New** spins up a container with sensible defaults and starts the agent inside it; multiple sessions share one per-case container; export a container + its workspace to a portable `.tar.gz` to move it to another machine. See [`docs/docker-cases.md`](docs/docker-cases.md) -- **Remote SSH sessions** — point a case at another machine and run the agent there inside a durable remote tmux: survives SSH drops, auto-reconnects, and can discover + attach sessions already running on the host. See [`docs/remote-sessions.md`](docs/remote-sessions.md) +- **Multi-CLI** — run **Claude Code**, **OpenCode**, **Codex**, **Antigravity**, **Gemini**, **Pi**, **Grok**, **DeepSeek Harness**, or **OMP** per session; env-var prefixes auto-gate (`CLAUDE_CODE_*` vs `OPENCODE_*` vs `CODEX_*` vs `ANTIGRAVITY_*` vs `GEMINI_*`/`GOOGLE_*` vs `PI_*` vs `GROK_*`/`XAI_*` vs `DSH_*`/`DEEPSEEK_*` vs `OMP_*`). See [`docs/opencode-integration.md`](docs/opencode-integration.md), [`docs/pi-integration.md`](docs/pi-integration.md), [`docs/grok-integration.md`](docs/grok-integration.md), [`docs/deepseek-integration.md`](docs/deepseek-integration.md) and [`docs/omp-integration.md`](docs/omp-integration.md) +- **Custom model endpoints** _(new in 1.29.0, HTTP API for now)_ — point a session's CLI at any OpenAI-compatible endpoint instead of its native backend: a local llama.cpp, llama-swap, Ollama or vLLM box, or a cloud gateway such as Azure AI Foundry or OpenRouter. Save an endpoint once (`POST /api/model-endpoints`; its models are discovered from `/v1/models`), apply it to a session (`POST /api/sessions/:id/custom-model`), and the CLI restarts in place on that endpoint. Verified live for Claude, OpenCode, Pi, Grok and OMP; Codex, Gemini and DeepSeek have documented gaps, Antigravity has no mechanism. A toolbar picker is the follow-up. See [`docs/custom-model-endpoints.md`](docs/custom-model-endpoints.md) +- **Web tabs** — open Grafana, Uptime Kuma, a Vite dev server or any dashboard URL as a tab beside your sessions (Run dropdown → **Web / URL** → **Add dashboard**). Dashboards are proxied through Codeman's own origin, so an `http://` target works from a phone over HTTPS and through the tunnel, single-page apps route on their own paths, and a frame that reloads recovers itself. A `localhost` link an agent prints opens as a web tab automatically. See [`docs/web-tabs.md`](docs/web-tabs.md) +- **Docker sessions** — run a case inside an isolated, hardened container. One checkbox on **Create New** spins up a container with sensible defaults and starts the agent inside it; multiple sessions share one per-case container, or attach a case to a container you already run; export a container + its workspace to a portable `.tar.gz` to move it to another machine. See [`docs/docker-cases.md`](docs/docker-cases.md) +- **Remote SSH sessions** — point a case at another machine and run the agent there inside a durable remote tmux: survives SSH drops, auto-reconnects, and can discover + attach sessions already running on the host; file previews and downloads come over the same ssh connection. See [`docs/remote-sessions.md`](docs/remote-sessions.md) - **Effort & Ultracode** — set a per-session default effort (`low`–`max`) or enable **ultracode** (dynamic multi-agent workflows). Soft defaults only — switchable anytime with `/effort` in-session. Extended-thinking budget is configurable too -- **Voice input** — dictate prompts with Deepgram Nova-3 (Web Speech API fallback): toggle recording, auto-silence stop, live level meter (`Ctrl+Shift+V`) +- **Voice input** — dictate prompts with Deepgram Nova-3, or through this machine's Claude Code login with no API key at all (App Settings → Voice; Web Speech API fallback): toggle recording, auto-silence stop, live level meter (`Ctrl+Shift+V`) - **Image input** — paste or drag-and-drop images straight into a session - **Gesture control** _(opt-in)_ — a MediaPipe hand-tracking overlay to grab/drag session windows and pinch buttons, hands-free. Enable with `CODEMAN_GESTURE=1` + App Settings → Terminal & Input - **Multi-monitor span** _(macOS)_ — one click opens a browser window maximized across all displays, so floating agent/gesture panels can cross the physical seam - **File Viewer button** _(opt-in)_ — a header button that toggles the built-in file browser panel with one tap; enable under App Settings → Header & Panels → Header buttons -- **CJK / IME input** — full composition support for Chinese / Japanese / Korean +- **CJK / IME input** — full composition support for Chinese / Japanese / Korean, with Ctrl- and Alt-modified navigation keys passed through to the CLI +- **Plan usage in the header** — live Claude subscription usage (the 5-hour and weekly windows) from a statusline exporter Codeman hands to `claude` at spawn and never writes into your settings files, plus Codex limits from its own app-server; per device, on for desktops and off for phones +- **Session list, your way** — the header strip, a left sidebar with a filter box, or a vertical rail whose detailed rows carry created and state stamps and sort by activity; the phone home screen and the desktop home rail use the same order +- **Terminal looks** — seven skins, four of them light, per-device font family and weight (the bundled JetBrains Mono covers weights 100 to 800), and opt-in entrance animations for tabs, agent windows, the terminal pane and connection lines - **OS notifications & hostname-aware titles** — desktop alerts and tab titles are prefixed `codeman:` so multi-host setups stay unambiguous --- @@ -461,8 +468,9 @@ Run a case inside its own hardened Docker container instead of directly on your - **Resource templates** — expand the checkbox for a **Small / Medium / Large / GPU** preset (memory, CPUs, GPU), or set your own. **Disk is elastic** — storage grows as data flows in, no fixed cap. - **Shared per-case container** — many sessions can `docker exec` into the same container; killing one session never tears the container out from under the others. - **Hardened by default** — non-root, `--cap-drop ALL`, `no-new-privileges`, PID/memory caps, never `--privileged` or the docker socket; a **sealed** profile (no host credentials, network off) is one toggle away. -- **Seamless auth, isolated credentials** — your host Claude / Codex / Antigravity / Gemini / OpenCode / Pi logins work inside the container out of the box: credentials are seeded (copied) in at launch and onboarding/trust prompts are pre-answered, so no login wizard appears. The container keeps its own copies and never writes back to your host credential stores; only conversation transcripts are shared, and exports never capture secrets. -- **Seamless auth, isolated credentials** — your host Claude / Codex / Antigravity / Gemini / OpenCode / OMP logins work inside the container out of the box: credentials are seeded (copied) in at launch and onboarding/trust prompts are pre-answered, so no login wizard appears. The container keeps its own copies and never writes back to your host credential stores; only conversation transcripts are shared, and exports never capture secrets.- **Move it to another machine** — export a container's whole environment (toolchain + workspace) to a portable `.tar.gz`, `docker load` it on the other side, and import it into a fresh case. +- **Seamless auth, isolated credentials** — your host Claude / Codex / Antigravity / Gemini / OpenCode / Pi / Grok / OMP logins work inside the container out of the box: credentials are seeded (copied) in at launch and onboarding/trust prompts are pre-answered, so no login wizard appears. The container keeps its own copies and never writes back to your host credential stores; only conversation transcripts are shared, and exports never capture secrets. +- **Attach to a container you already run** — tick **Attach to an existing container** on the Docker panel to link a case to it instead of creating one. Codeman only `exec`s into it and never starts, stops, restarts or removes it; one adopted container can back several cases at different directories, and **copy an existing case** pre-fills the form from a sibling. Admin-only in multi-user mode, since the container's mounts belong to whoever started it. +- **Move it to another machine** — export a container's whole environment (toolchain + workspace) to a portable `.tar.gz`, `docker load` it on the other side, and import it into a fresh case. - **Durable** — reconnect after a restart lands back in the same live agent; a container stop/reboot resumes the conversation from the bind-mounted transcript. Prerequisite: just Docker (or Podman). The agent base image builds itself automatically on first use, with progress streamed to the UI (or pre-build it with `node scripts/build-agent-image.mjs`). Full guide: [`docs/docker-cases.md`](docs/docker-cases.md). @@ -478,6 +486,7 @@ Point a case at another machine and run the agent **there**, over SSH, with the - **Discover & attach**: list the `codeman-*` sessions already running on a host (started by that machine's own Codeman, or by another operator) and attach to one. Attached sessions you don't own **detach on tab close, never kill**. - **Shared sessions**: several clients can attach the same remote session at different window sizes without clamping each other; discovery shows a "shared" badge with the client count. - **Injection-safe**: every ssh command line flows through a single shell-escaping builder, and host/path/identity fields are schema-guarded. +- **Files too**: previews, downloads and text reads in a remote case go over the same ssh connection (one `realpath` + `stat` probe, then a streamed `cat`, `Range` seeking included), so a clicked path opens the file on the machine the agent is on. Nothing is copied to the Codeman host; editing and Office previews answer a clear 400 instead of a misleading 404. Set it up under **New Case → Remote** (host, user, identity file, optional jump host). Full design: [`docs/remote-sessions.md`](docs/remote-sessions.md). @@ -647,8 +656,8 @@ These run for **every** request — before auth, even on the default no-password ### Input, files & headers -- **Schema-validated inputs** — every API body is checked with Zod v4 schemas; a `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `ANTIGRAVITY_*` / `GEMINI_*` / `GOOGLE_*` / `PI_*` env-prefix allowlist gates which settings each CLI can receive -- **Path containment** — file routes `realpath` before boundary checks (no TOCTOU); `..`, absolute paths, and symlinks resolving outside the working dir are rejected. Caps: 10 MB text preview / 50 MB raw & download; `/api/download` blocklists sensitive paths (`.env`, `*credentials*`, `~/.ssh/`, `.aws/credentials`). SVG/HTML is served `octet-stream` + `nosniff` + attachment so it downloads rather than executes +- **Schema-validated inputs** — every API body is checked with Zod v4 schemas; a `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `ANTIGRAVITY_*` / `GEMINI_*` / `GOOGLE_*` / `PI_*` / `GROK_*` / `XAI_*` / `DSH_*` / `DEEPSEEK_*` / `OMP_*` env-prefix allowlist gates which settings each CLI can receive, and the keys that could redirect a CLI's traffic (base URLs, config homes) are clamped for non-admin users +- **Path containment** — file routes `realpath` before boundary checks (no TOCTOU); `..`, absolute paths, and symlinks resolving outside the working dir are rejected. Caps: 10 MB text preview / 2 GB raw & download (`CODEMAN_MAX_DOWNLOAD_BYTES`; bodies stream and answer `Range` requests, so the cap is a sanity bound rather than memory protection); `/api/download` blocklists sensitive paths (`.env`, `*credentials*`, `~/.ssh/`, `.aws/credentials`). SVG/HTML is served `octet-stream` + `nosniff` + attachment so it downloads rather than executes - **Security headers** — `Content-Security-Policy` (`default-src 'self'`, every exception enumerated), `X-Content-Type-Options: nosniff`, `X-Frame-Options: SAMEORIGIN`, HSTS over HTTPS, and CORS reflected **only** for `localhost` / `127.0.0.1` / `::1` ### Supply chain & isolation @@ -698,6 +707,10 @@ The web UI remains the primary surface; see **[docs/tui.md](docs/tui.md)** for t | `Ctrl/Cmd +` / `-` | Font size | | `Ctrl/Cmd+?` | Keyboard help | | `Shift+Enter` | Insert newline (sent to terminal) | +| `Shift+drag` | Select text in a pane whose mouse events go to the CLI | +| Right-click | Copy the selection (the native menu stays when nothing is selected) | +| `Shift+Wheel` | Scroll the local scrollback while the wheel is forwarded to the CLI | +| `Ctrl+Z` | Swallowed in agent sessions so a running CLI cannot be suspended; normal job control in a shell | | `Escape` | Close panels & modals | --- @@ -762,7 +775,7 @@ Those `DONE__` strings are the skill's **split marker** trick, and | --------------------------------------------------------------------- | --------------------------------------------------------------------------------------------- | | [`SKILL.md`](skills/codeman/SKILL.md) | Safety rules, the ready-made fast path (spawn N workers, task them, collect), and the verb index. Always loaded. | | [`reference/verbs.md`](skills/codeman/reference/verbs.md) | The 14 verbs in detail: readiness, send-and-wait, markers, interrupts, cleanup. On demand. | -| [`reference/recipes.md`](skills/codeman/reference/recipes.md) | 6 worked multi-worker flows (fan-out, blocked-worker watch, messaging fan-out). On demand. | +| [`reference/recipes.md`](skills/codeman/reference/recipes.md) | 8 worked flows: claude, DeepSeek Harness and shell workers, fan-out, blocked-worker watch, messaging fan-out. On demand. | | [`reference/endpoints.md`](skills/codeman/reference/endpoints.md) | Full endpoint tables, error codes, per-mode signal table, capacity limits. On demand. | | [`reference/messaging.md`](skills/codeman/reference/messaging.md) | Talking to claude workers directly via Claude Code cross-session messaging. On demand. | @@ -798,8 +811,8 @@ When a CLI runs in a Codeman-managed session, these environment variables are se 4. **Response envelope.** Most endpoints return `{ "success": true, "data": … }` (errors: `{ "success": false, "error", "errorCode" }`). A few legacy GETs return bare bodies — **handle both** (`body.data ?? body`). 5. **`/api/v1/*`** is a stable alias of `/api/*`. 6. **Wait instead of polling, and don't treat a timeout as an error.** The wait endpoints answer with HTTP `200` and `wait.timedOut: true` when nothing happened in time, so loop over short waits (60s is the default) rather than issuing one long call, because tunnels cut idle connections. `wait.timeoutMs` tells you the timeout the server actually applied after clamping (600s ceiling). -7. **Only `claude` sessions emit `stop` and `blocked`.** Those two come from Claude Code hooks; `shell` and the external CLIs (opencode/codex/gemini/antigravity/pi) accept only `idle`, `working` and `exit`. Asking for `stop` explicitly on those is a `400`; omitting `until` is always safe. ⚠️ On a `shell` session `idle` fires **once**, at startup, and never again, so send-and-wait there can only time out; synchronize hook-less sessions with a `wait-output` marker. -7. **Only `claude` sessions emit `stop` and `blocked`.** Those two come from Claude Code hooks; `shell` and the external CLIs (opencode/codex/gemini/antigravity/omp) accept only `idle`, `working` and `exit`. Asking for `stop` explicitly on those is a `400`; omitting `until` is always safe. ⚠️ On a `shell` session `idle` fires **once**, at startup, and never again, so send-and-wait there can only time out; synchronize hook-less sessions with a `wait-output` marker.8. **Nothing reports "ready", so wait for it explicitly.** A new session answers `{"signal":"exit","immediate":true}` (that means *not started*, not *crashed*) until its PID exists, and a `claude` worker in a fresh case then sits on the CLI's trust dialog. Prompt it there and the wait resolves on `idle` in ~2s looking exactly like a finished turn, while the text sits stuck in the dialog. Recipe 2b below is the sequence that avoids it. +7. **Only `claude` and `deepseek` sessions emit `stop` and `blocked`.** Those two come from hooks (Claude Code's own, and the DeepSeek Harness status bridge); `shell` and the other external CLIs (opencode/codex/gemini/antigravity/pi/grok/omp) accept only `idle`, `working` and `exit`. Asking for `stop` explicitly on those is a `400`; omitting `until` is always safe. ⚠️ On a `shell` session `idle` fires **once**, at startup, and never again, so send-and-wait there can only time out; synchronize hook-less sessions with a `wait-output` marker. +8. **Nothing reports "ready", so wait for it explicitly.** A new session answers `{"signal":"exit","immediate":true}` (that means *not started*, not *crashed*) until its PID exists, and a `claude` worker in a fresh case then sits on the CLI's trust dialog. Prompt it there and the wait resolves on `idle` in ~2s looking exactly like a finished turn, while the text sits stuck in the dialog. Recipe 2b below is the sequence that avoids it. ### Recipes @@ -866,9 +879,20 @@ curl -sG "$API/api/sessions/$SID/wait-output" \ --data-urlencode "match=DONE_$N" --data-urlencode 'from=buffer' \ --data-urlencode 'timeout=60000' | jq '.data.wait' -# 5. Read the terminal back. ⚠️ Use terminal?tail=, NOT /output: the latter's -# textOutput is empty for every tmux-backed (i.e. every interactive) session. -# tail counts BYTES, and what comes back is terminal data, ANSI included. +# 5. Read the answer. claude / codex / deepseek sessions have last-response: it comes +# from the transcript, not the screen, so no TUI frames or repaint noise. +# ⚠️ Poll rather than read once: the transcript lands slightly after the stop +# signal, so a read right after send-and-wait returns often comes back empty. +for _ in $(seq 1 10); do + TXT=$(curl -s "$API/api/sessions/$SID/last-response" | jq -r '.data.text') + [ -n "$TXT" ] && break; sleep 1 +done +printf '%s\n' "$TXT" + +# 5b. Other modes (shell/opencode/gemini/antigravity/pi/grok/omp) have no transcript: +# read the terminal. ⚠️ Use terminal?tail=, NOT /output: the latter's textOutput +# is empty for every tmux-backed (i.e. every interactive) session. tail counts +# BYTES, and what comes back is terminal data, ANSI included. curl -s "$API/api/sessions/$SID/terminal?tail=8000" | jq -r '.data.terminalBuffer' # 6. Stream live events (session output, agent activity, status) @@ -914,7 +938,7 @@ Codeman registers Claude Code hooks that `POST /api/hook-event` (`permission_pro ## API -REST over Fastify — **~200 handlers across 21 route modules**, plus an SSE stream and a WebSocket terminal channel. All responses use the `ApiResponse` envelope (`{success, data}` / `{success, error, errorCode}`); `/api/v1/*` is a stable alias. A representative subset: +REST over Fastify — **~230 handlers across 25 route modules**, plus an SSE stream and a WebSocket terminal channel. All responses use the `ApiResponse` envelope (`{success, data}` / `{success, error, errorCode}`); `/api/v1/*` is a stable alias. A representative subset: ### Sessions @@ -925,11 +949,13 @@ REST over Fastify — **~200 handlers across 21 route modules**, plus an SSE str | `POST` | `/api/sessions/:id/input` | Send input (`{input, useMux?, clientId?, seq?, wait?, waitTimeout?}`: `clientId`+`seq` = exactly-once; `wait` blocks until the turn ends) | | `GET` | `/api/sessions/:id/terminal` | Read terminal output (`?tail=`, `?full=1`); the read path for interactive sessions | | `GET` | `/api/sessions/:id/output` | Parsed one-shot output (`textOutput` is empty for tmux-backed sessions) | +| `GET` | `/api/sessions/:id/last-response` | The last answer as clean text, read from the transcript (claude, codex, deepseek) | | `GET` | `/api/sessions/:id/wait` | Block until a signal fires (`?until=stop,idle,exit&timeout=&fresh=`); a timeout is a `200` | | `GET` | `/api/sessions/:id/wait-output` | Block until a literal string appears (`?match=&nocase=&from=now\|buffer&timeout=`) | | `GET` | `/api/sessions/unified` | Unified live + history list (Session Manager) — `?q=&limit=` | | `POST` | `/api/sessions/:id/pin` | Pin/unpin in the Session Manager (`{pinned}`) | | `PUT` | `/api/session-order` | Sync tab order across devices (`{order: [ids]}`) | +| `POST` | `/api/sessions/:id/custom-model` | Restart the session's CLI on a saved custom endpoint (`{endpointId, modelId}`; `{clear: true}` returns to the native backend) | | `DELETE` | `/api/sessions/:id` | Delete session | ### Respawn @@ -978,6 +1004,7 @@ REST over Fastify — **~200 handlers across 21 route modules**, plus an SSE str | `GET` | `/api/system/update/check` | Check for a new release | | `POST` | `/api/system/update` | Self-update (git-clone installs) | | `POST` | `/api/clipboard` | Push text to all connected browsers (`{text}`) | +| `GET` / `POST` | `/api/model-endpoints` | List / save custom OpenAI-compatible endpoints (`PUT` / `DELETE` `/:id`; admin-only in multi-user mode) | | `GET` | `/api/sessions/:id/run-summary` | Timeline + stats | > **Building something on top of Codeman?** [`docs/extending-codeman.md`](docs/extending-codeman.md) is the integration guide: render your own UI as a tab, subscribe to the SSE event stream to react when an agent needs you, drive Codeman from a script, and the traps worth knowing before you start. Codeman has no plugin runtime on purpose, so an integration is just your own process talking HTTP. @@ -1014,8 +1041,8 @@ flowchart TB end subgraph External["External"] - CLI["AI CLI
Claude Code / OpenCode / Codex / Antigravity / Gemini / Pi"] - CLI["AI CLI
Claude Code / OpenCode / Codex / Antigravity / Gemini / OMP"] BG["Background Agents
(Task tool)"] + CLI["AI CLI
Claude Code / OpenCode / Codex / Antigravity / Gemini / Pi / Grok / DeepSeek / OMP"] + BG["Background Agents
(Task tool)"] end end @@ -1081,7 +1108,7 @@ Full details: [`docs/archive/code-structure-findings.md`](docs/archive/code-stru [![npm](https://img.shields.io/npm/v/xterm-zerolag-input?style=flat-square&color=22c55e)](https://www.npmjs.com/package/xterm-zerolag-input) -Instant keystroke feedback overlay for xterm.js. Eliminates perceived input latency over high-RTT connections by rendering typed characters immediately as a pixel-perfect DOM overlay. Zero dependencies, 6.1 kB gzipped, configurable prompt detection, CJK/emoji wide-character support, full state machine with 175 tests. +Instant keystroke feedback overlay for xterm.js. Eliminates perceived input latency over high-RTT connections by rendering typed characters immediately as a pixel-perfect DOM overlay. Zero dependencies, 6.1 kB gzipped, configurable prompt detection, CJK/emoji wide-character support, full state machine with 238 tests. ```bash npm install xterm-zerolag-input diff --git a/README.zh-CN.md b/README.zh-CN.md index 966884ba..8f19c042 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -5,7 +5,7 @@

AI 编程智能体的任务控制中心

- Claude Code • OpenCode • Codex • Antigravity • Gemini • Pi • Grok • 终端 —— 统一仪表盘 • 任意设备 + Claude Code • OpenCode • Codex • Antigravity • Gemini • Pi • Grok • DeepSeek • OMP • 终端 —— 统一仪表盘 • 任意设备

@@ -17,6 +17,8 @@ Node.js 22+ TypeScript 5.9 Fastify + npm version + GitHub stars Contributors Total commits

@@ -25,12 +27,10 @@ Codeman — 并行子智能体可视化

-

- Codeman 仪表盘导览:按项目分组的会话标签页、一键 Run 启动新智能体、页头实时用量 -

- > 本文档由英文版 [`README.md`](README.md) 翻译而来。如有出入,以英文版为准。 +**Codeman** 是一个自托管的 AI 编程智能体任务控制中心。它在持久化的 tmux 会话里拉起 Claude Code、OpenCode、Codex、Antigravity、Gemini、Pi、Grok、DeepSeek Harness 或 OMP,把真实的终端流式传到任意浏览器,并在你离开之后让智能体继续干活:空闲时重新提示、用量限额重置后自动续跑、按计划执行任务,还能实时展示每一个后台智能体的工作。 + 一行命令即可安装(macOS 和 Linux,Windows 通过 WSL): ```bash @@ -44,6 +44,17 @@ codeman web 安装器在每次系统改动前都会先询问;重跑同一条命令即可原地更新。详见[快速开始 — 安装](#快速开始--安装)。 +- **一个仪表盘,九个 CLI**:每个会话可选 [Claude Code、OpenCode、Codex、Antigravity、Gemini、Pi、Grok、DeepSeek 或 OMP](#更多特性)(外加普通 shell),在本机、[Docker 容器](#隔离的-docker-会话)或 [SSH 远程主机](#远程-ssh-会话)上运行,你自己的仪表盘也能作为 [Web 标签页](#更多特性)并排打开 +- **真正的手机友好**:[触控优化的终端](#移动端优化的-web-ui),即时本地回显、二维码登录、滑动导航与推送通知 +- **睡觉时也在跑**:[空闲检测 + 重生循环](#重生控制器respawn-controller),订阅限额重置后自动续跑,支持 24 小时以上的无人值守运行 +- **看见智能体在想什么**:每个子智能体和团队成员都有[实时浮动窗口](#实时智能体可视化),附带实时活动记录 +- **什么都不会丢**:tmux 让会话挺过重启和断网,输入精确一次送达,完整的回滚缓冲区回放 +- **自托管、私有**:默认仅环回、MIT 许可、无遥测,完全运行在你自己的机器上 + +

+ Codeman 仪表盘导览:按项目分组的会话标签页、一键 Run 启动新智能体、页头实时用量 +

+ --- ## 快速开始 — 安装 @@ -52,13 +63,14 @@ codeman web curl -fsSL https://getcodeman.com/install | bash ``` -该脚本会在缺失时自动安装 Node.js 和 tmux,把 Codeman 克隆到 `~/.codeman/app` 并完成构建。几点须知: +该脚本会在缺失时自动安装 Node.js、tmux 和一套构建工具链(node-pty 没有 Linux 预编译包,需要从源码编译),把 Codeman 克隆到 `~/.codeman/app` 并完成构建。几点须知: - **先询问,后改动。** 所有系统级改动(安装软件包、下载 AI CLI)都会先征求确认;结束时的菜单可选择:直接在本终端运行、安装为后台服务(systemd/launchd,开机自启),或暂不启动。不选就不会有任何后台进程。 +- **怎么访问,由你决定。** 安装器提供三种到达仪表盘的方式:**Tailscale**(环回绑定,由 `tailscale serve` 代理,得到带真实证书的 `https://<机器名>..ts.net`,用你的 tailnet 当登录,无需密码)、**局域网内任意设备**(`0.0.0.0`,会提示设置一个强烈推荐的密码),或**仅本机**(`127.0.0.1`,最安全)。绑定网络却跳过密码需要显式确认,并以醒目警告收尾。高亮的默认项反映机器上已有的状态(已在用 Tailscale 时默认 Tailscale,重跑时沿用现有绑定),直接回车绝不会引入新软件。手动运行的 `codeman web` 仍默认仅环回。 - **重跑即更新。** 再次运行同一条命令即可原地更新已完成的安装:`~/.codeman/app` 中的本地改动会被 stash(绝不丢弃),运行中的服务会自动重启并校验。若首次安装中途失败,重跑会继续完成完整的安装流程。也可以使用 `install.sh update` 与 `install.sh uninstall`。 - **CI / 无终端环境:** 没有终端时,涉及系统改动的步骤会带着说明中止,而不是静默执行;在自动化场景设置 `CODEMAN_NONINTERACTIVE=1` 即可批准这些步骤。 -你至少需要安装一个 AI 编程 CLI —— [Claude Code](https://docs.anthropic.com/en/docs/claude-code)、[OpenCode](https://opencode.ai)、[Codex](https://developers.openai.com/codex/cli)、[Antigravity](https://antigravity.google)、[Gemini CLI](https://github.com/google-gemini/gemini-cli)、[Pi](https://pi.dev)、[Grok Build](https://github.com/xai-org/grok-build)、[DeepSeek Harness](https://github.com/deepseek-ai/deepseek-harness) 或 [OMP](https://github.com/can1357/oh-my-pi)(任意组合均可;自 Google 面向消费者停售后,Gemini CLI 仅限企业版,Antigravity 是其继任者)。安装器会自动检测这九个中已安装的任意一个;若一个都没有,会提供安装 Claude Code 或 OpenCode 的选项,也可以选择跳过、稍后自行安装。安装完成后: +你至少需要安装一个 AI 编程 CLI —— [Claude Code](https://docs.anthropic.com/en/docs/claude-code)、[OpenCode](https://opencode.ai)、[Codex](https://developers.openai.com/codex/cli)、[Antigravity](https://antigravity.google)、[Gemini CLI](https://github.com/google-gemini/gemini-cli)、[Pi](https://pi.dev)、[Grok Build](https://github.com/xai-org/grok-build)、[DeepSeek Harness](https://github.com/deepseek-ai/deepseek-harness) 或 [OMP](https://github.com/can1357/oh-my-pi)(任意组合均可;自 Google 面向消费者停售后,Gemini CLI 仅限企业版,Antigravity 是其继任者)。安装器会自动检测这九个中已安装的任意一个;若一个都没有,会给出一个菜单让你安装其中任意一个(DeepSeek 除外,它的 npm 包只装一个启动器,没有可运行的 profile),也可以选择跳过、稍后自行安装。安装完成后: ```bash codeman web @@ -72,12 +84,34 @@ codeman users add alice --admin # 创建第一个管理员账号 codeman web --multiuser # 命名登录 + 按用户隔离的案例空间 ``` +**更喜欢 Docker Compose?** `docker/` 里附带一套本地镜像的 Compose 部署:把 `docker/.env.example` 复制为 `docker/.env`,设置 `CODEMAN_PASSWORD`,然后在 Linux 上运行 `bash docker/Start-Codeman.sh`。Codeman 自己跑在容器里,并通过宿主机的 socket 把 Docker 案例作为并列容器拉起。更新之后请再跑一次这个脚本,而不是直接 `docker compose up`,这样重建的镜像、刷新的卷和新的入口脚本会一起就位。直接的 Compose 命令、存储与网络选项见 [Docker 部署指南](docker/README.md)(英文)。 + 详见下文[多用户模式](#多用户模式可选启用)。
-作为后台服务运行 +让它在后台一直运行 -安装器结尾的菜单(选项 2)可以帮你完成这一步,并在宣告成功前校验服务确实已启动。如需手动配置: +想让它活过你启动它的那个 shell,而且什么都不用配置: + +```bash +codeman web -d # 脱离终端;日志写到 ~/.codeman/web.log +codeman web --status # 是否在运行,pid 是多少 +codeman web --stop # 优雅的 SIGTERM;智能体继续留在 tmux 里运行 +``` + +`-d` 会等到服务器真正应答后才报告成功,并且拒绝在同一个数据目录上启动第二个(两个服务器共用一个 tmux socket 会互相附着对方的会话)。 + +想让它在重启后自动回来,就装成服务。安装器结尾的菜单(选项 2)会替你完成;`codeman service` 是 `npm i -g aicodeman` 安装的等价物: + +```bash +codeman service install # systemd 用户单元(Linux)或 LaunchAgent(macOS) +codeman service status +codeman service uninstall +``` + +`service install` 会把你当前的 PATH 写进单元文件,这比听起来重要得多:launchd 只给任务 `/usr/bin:/bin:/usr/sbin:/sbin`,所以手写的 plist 根本找不到 Homebrew 或 nvm 装的 `node`、`tmux` 或 `claude`。它绝不会把 `CODEMAN_PASSWORD` 复制进单元文件;服务需要认证的话请自行添加。 + +如需手动编写单元文件: **Linux(systemd):** @@ -177,17 +211,17 @@ Codeman 依赖 tmux,因此 Windows 用户需要 [WSL](https://learn.microsoft. 在手机上手打密码扫二维码 —— 即时认证 -- **键盘配件栏** —— 在虚拟键盘上方提供 `/init`、`/clear`、`/compact` 快捷按钮;破坏性命令需双击确认,绝不误触 +- **键盘配件栏** —— 在虚拟键盘上方提供 `/init`、`/clear`、`/compact` 快捷按钮;破坏性命令需双击确认,绝不误触;在 Codex 会话上还会显示 `⇧←` / `⇧→`(Shift+Left / Shift+Right:编辑上一条排队的消息 / 在提示栈里回退) - **独立的 Enter 按钮** —— 以按键方式回放,先冲刷本地回显缓冲的文本,不会让内容滞留在屏幕上 - **滑动导航与智能键盘处理** —— 左右滑动切换会话;键盘弹出时工具栏与终端整体上移(`visualViewport` API) -- **为手机而生** —— 刘海与 Home 指示条的安全区适配、44px 触控目标、底部抽屉式 case 选择器、原生惯性滚动 +- **为手机而生** —— 刘海与 Home 指示条的安全区适配、44px 触控目标、底部抽屉式 case 选择器、原生惯性滚动;折叠屏手机(iPhone Duo)上对话框会避开铰链,开合设备也绝不会被误判成键盘弹出 ```bash codeman web --https # 在手机上打开:https://<你的IP>:3000 ``` -> `localhost` 走纯 HTTP 即可。从其他设备访问时请使用 `--https`,或使用 [Tailscale](https://tailscale.com/)(推荐)—— 它提供私有网络,让你无需 TLS 证书即可从手机访问 `http://:3000`。 +> `localhost` 走纯 HTTP 即可。从其他设备访问时请使用 `--https`,或使用 [Tailscale](https://tailscale.com/)(推荐):安装器可以替你配好(在网络访问提示处选择 **Tailscale**,或在已有安装上运行 `bash ~/.codeman/app/install.sh tailscale`)。这样你会得到带真实证书的 `https://<你的机器>..ts.net`:只对你的 tailnet 可见、无需密码,手机上的 PWA 安装和推送通知也都能用。 ### 安全的二维码认证 @@ -210,6 +244,8 @@ codeman web # localhost:3000(仅环回 —— 安全默 codeman web --port 8080 # 自定义端口(或设置 CODEMAN_PORT) codeman web --https # 自签名 TLS(仅远程访问时需要) codeman web -H 0.0.0.0 # 绑定局域网 —— 必须设置 CODEMAN_PASSWORD(见「安全」) +codeman web -d # 脱离终端:关掉 shell 也在跑(--status、--stop) +codeman service install # systemd/launchd 服务:重启后自动回来 ``` 打开打印出的 URL。整个页面是一个单一仪表盘;下面的一切都在这里完成。 @@ -220,16 +256,16 @@ codeman web -H 0.0.0.0 # 绑定局域网 —— 必须设置 CODEMAN_ | 字段 | 作用 | | ---------------------- | ------------------------------------------------------------------------------------------- | -| **工作目录 / case** | 智能体操作的文件夹。「case」就是一个 Codeman 记住的命名工作目录。 | -| **CLI / 运行模式** | `Claude`(默认)、`OpenCode`、`Codex`、`Antigravity`、`Gemini`、`Pi`、`Grok` 或 `Terminal`(普通 shell)。 | -| **模型** | 每会话模型(App Settings → Claude Model)。软默认值 —— 会话内 `/model` 依然有效。 | +| **工作目录 / case** | 智能体操作的文件夹。「case」就是一个 Codeman 记住的命名工作目录。**Add Case** 可以从零创建、链接一个已有文件夹,或把一个 GitHub 仓库直接克隆成 case(**Clone Repo**)。 | +| **CLI / 运行模式** | `Claude`(默认)、`OpenCode`、`Codex`、`Antigravity`、`Gemini`、`Pi`、`Grok`、`DeepSeek`、`OMP` 或 `Terminal`(普通 shell)。 | +| **模型** | 每会话模型(App Settings → Models → New Claude sessions)。软默认值 —— 会话内 `/model` 依然有效。 | | **Effort / Ultracode** | 推理力度(`low`–`max`),或用 `ultracode` 开启动态多智能体工作流。随时可用 `/effort` 切换。 | 点击启动 —— Codeman 通过真实 PTY 拉起 CLI,并经 SSE 流式传输到你的浏览器。 ### 3. 读懂仪表盘 -- **标签(顶部)** —— 每个会话一个。`Alt+1`–`9` 跳转,`Ctrl+Tab` 下一个,拖拽排序(标签顺序会跨设备同步)。 +- **标签(顶部)** —— 每个会话一个。`Alt+1`–`9` 跳转,`Ctrl+Tab` 下一个,拖拽排序(标签顺序会跨设备同步)。更喜欢列表?**App Settings → Appearance → Tabs** 可以把它挪进左侧边栏(带筛选框,`Alt+B` 折叠)或一条竖向导轨,导轨的行按活动状态排序:先是等你处理的,然后是跑得最久的,最后是刚刚安静下来的。 - **终端(中央)** —— 真实的 `xterm.js` 终端;完整 TUI 正常渲染。直接输入并按 **Enter** 发送。`Shift+Enter` 插入换行。 - **侧边面板** —— Respawn、Orchestrator、Cron、Subagents、Settings(从工具栏切换)。 @@ -237,8 +273,10 @@ codeman web -H 0.0.0.0 # 绑定局域网 —— 必须设置 CODEMAN_ - **直接在终端输入提示** —— 即使跨越重连,输入也是精确一次送达(连接中断绝不会丢失或重复发送提示)。 - **粘贴或拖放图片**,直接进入会话。 -- **语音输入** —— `Ctrl+Shift+V`(Deepgram Nova-3,自动静音停止)。 -- **附件** —— 注册外部文件/文档,并内联预览 Office/PDF。 +- **语音输入** —— `Ctrl+Shift+V`(Deepgram Nova-3,或者直接用这台机器的 Claude Code 登录、不需要任何 API key;自动静音停止)。 +- **附件** —— 注册外部文件/文档,并内联预览 Office/PDF;智能体打印出的任何文件路径都可以点击,终端里和对话视图里都行。 +- **需要你的时候** —— 标签会变黄(等待输入)或变红(有个问题挡住了它)。**审批收件箱(Approvals Inbox)**(可选启用)把所有会话里等着你的提示排成一个队列,可以从页头的铃铛或手机首页直接作答;🧠 **Read My Mind**(可选启用)会根据这个 case 的目标和最近的工作替你起草下一条提示。 +- **看到什么就能复制什么** —— `Shift+拖动` 在 CLI 接管了鼠标时也能选中文本,右键复制选中内容,自动复制(Auto Copy,可选启用)在松开鼠标的瞬间就复制。 ### 5. 让它自主运行 @@ -246,7 +284,7 @@ codeman web -H 0.0.0.0 # 绑定局域网 —— 必须设置 CODEMAN_ | ---------------- | --------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------ | | **Respawn** | 长时间无人值守运行 —— 空闲/限额时自动重启 CLI,带自适应时序。预设:`solo-work`、`overnight-autonomous` 等 | Respawn 标签页 | | **Orchestrator** | 把一个目标变成分阶段计划,并跨多个智能体推动完成。 | 编排器面板 | -| **Cron** | 已保存的、命名的定时任务(`once`/`interval`/`daily`/`weekly`),到期时拉起会话并发送提示。 | ⏰ Cron 按钮(可选启用:App Settings → Display → Header Displays) | +| **Cron** | 已保存的、命名的定时任务(`once`/`interval`/`daily`/`weekly`),到期时拉起会话并发送提示。 | ⏰ Cron 按钮(可选启用:App Settings → Header & Panels → Scheduling) | | **Auto-resume** | 订阅限额重置后自动继续。 | Respawn 标签页(顶部) | ### 6. 随时随地访问 @@ -257,8 +295,9 @@ codeman web -H 0.0.0.0 # 绑定局域网 —— 必须设置 CODEMAN_ ### 7. 运维与维护 -- **App Settings** —— 模型、effort、权限启动模式、主题/皮肤、通知、显示开关、各 CLI 的专属选项,以及跨设备同步的自定义显示名称和按设备保存的英文/简体中文界面语言。 -- **自更新** —— git-clone 安装可在 **Settings → Updates** 中原地更新。 +- **App Settings** —— 模型、effort、权限启动模式、主题/皮肤、终端字体与字重、入场动画、通知、显示开关、各 CLI 的专属选项,以及跨设备同步的自定义显示名称和按设备保存的英文/简体中文界面语言。 +- **让它在后台运行** —— `codeman web -d` 脱离你的 shell(`--status`、`--stop`);`codeman service install` 把它装成 systemd 用户单元 / macOS LaunchAgent,重启后自动回来。两者都会先确认服务器真正应答再报告成功,也都拒绝在同一个数据目录上启动第二个服务器。见[让它在后台一直运行](#快速开始--安装)。 +- **自更新** —— git-clone 安装可在 **App Settings → System → Updates** 中原地更新。 - **部署你自己的改动** —— 见[开发](#开发)。 > ⚠️ **安全提示:** 如果你正在 Codeman 受管会话*内部*工作(`echo $CODEMAN_MUX` → `1`),绝不要直接运行 `tmux kill-session` / `pkill claude` —— 请使用 Web UI 或 `./scripts/tmux-manager.sh`。 @@ -373,6 +412,14 @@ codeman web --title-hostname dev-box # codeman:dev-box(用于覆盖嘈 | **110k tokens** | 自动 `/compact` | 上下文被摘要,工作继续 | | **140k tokens** | 自动 `/clear` | 以 `/init` 全新开始 | +### 标签提醒(Tab Alerts) + +

+ 会话标签:一个普通的活动标签,旁边是黄色的等待输入标签和红色的需要决定标签,都带着呼吸式光晕 +

+ +每个标签一眼就能看出状态。运行中的会话保持绿色状态点。会话停下来等待输入时,标签变**黄**:稳定的描边、着色的背景、黄色的点,上面叠一层缓慢的呼吸光晕。当权限提示或提问**挡住**了智能体,标签变**红**,脉动更快。底色永远不会闪灭,所以哪怕只瞥一眼(或截一张图)也能读到真实状态;标签被选中时描边依然可见,页面刷新后会从服务端重新装载待处理的提醒,因此一个被挡住的会话绝不可能藏在一个看起来正常的标签后面。 + ### 通知 当会话需要关注时实时桌面提醒 —— `permission_prompt` 与 `elicitation_dialog` 触发关键的红色标签闪烁,`idle_prompt` 触发黄色闪烁。点击任意通知即可直接跳转到相关会话。Hook 按 case 目录自动配置。 @@ -393,17 +440,24 @@ PTY 输出 → 16ms 服务端批处理 → DEC 2026 包裹 → SSE → 客户端 ## 更多特性 -- **自更新** —— systemd/launchd 管理下的 git-clone 安装可在 **App Settings → Updates** 中原地更新:它会检测最新发行版,自动暂存(stash)脏工作树,并在服务重启期间流式展示构建进度(npm 安装会被报告为不可更新) -- **多 CLI** —— 每个会话可选 **Claude Code**、**OpenCode**、**Codex**、**Antigravity**、**Gemini**、**Pi** 或 **Grok**;环境变量前缀自动隔离(`CLAUDE_CODE_*`、`OPENCODE_*`、`CODEX_*`、`ANTIGRAVITY_*`、`PI_*`、`GROK_*`/`XAI_*` 与 `GEMINI_*`/`GOOGLE_*`)。详见 [`docs/opencode-integration.md`](docs/opencode-integration.md)、[`docs/pi-integration.md`](docs/pi-integration.md) 与 [`docs/grok-integration.md`](docs/grok-integration.md) -- **Docker 会话** —— 在隔离且加固的容器中运行案例。**Create New** 上勾选一个复选框即可用合理的默认值启动容器并在其中启动智能体;同一案例的多个会话共享一个容器;可将容器连同工作区导出为可移植的 `.tar.gz`,迁移到另一台机器。详见 [`docs/docker-cases.md`](docs/docker-cases.md) -- **远程 SSH 会话**:把案例指向另一台机器,让智能体在那里一个持久的远程 tmux 中运行:SSH 断连不中断任务、自动重连,还能发现并附着主机上已在运行的会话。详见 [`docs/remote-sessions.md`](docs/remote-sessions.md) +- **后台守护进程与服务安装** —— `codeman web -d` 以脱离终端的方式运行服务器,带 pid 文件、`~/.codeman/web.log` 和经过校验的启动(它会轮询到服务器应答为止,所以端口冲突绝不会被当成成功);`codeman service install` 写入一个 systemd 用户单元(Linux)或 LaunchAgent(macOS),并把你 shell 的 PATH 一并写进去,这样 nvm 或 Homebrew 装的 `node`、`tmux` 和 `claude` 才真的找得到。机密永远不会写进单元文件 +- **自更新** —— systemd/launchd 管理下的 git-clone 安装可在 **App Settings → System → Updates** 中原地更新:它会检测最新发行版,自动暂存(stash)脏工作树,并在服务重启期间流式展示构建进度(npm 安装会被报告为不可更新) +- **把 GitHub 仓库克隆成 case** —— 在 **Add Case → Clone Repo** 里粘贴一个仓库 URL,Codeman 会把它克隆到 `~/codeman-cases/` 并注册为普通 case,随时可以跑智能体。输入时它会预检 URL(告诉你能否匿名克隆,并为可选的分支/标签字段提供仓库真实的分支与标签),从 URL 里填好 case 名,还让你选 Run 按钮该用哪个 CLI。支持 `https://` 的公开仓库;Codeman 绝不收集或保存凭据 +- **多 CLI** —— 每个会话可选 **Claude Code**、**OpenCode**、**Codex**、**Antigravity**、**Gemini**、**Pi**、**Grok**、**DeepSeek Harness** 或 **OMP**;环境变量前缀自动隔离(`CLAUDE_CODE_*`、`OPENCODE_*`、`CODEX_*`、`ANTIGRAVITY_*`、`GEMINI_*`/`GOOGLE_*`、`PI_*`、`GROK_*`/`XAI_*`、`DSH_*`/`DEEPSEEK_*` 与 `OMP_*`)。详见 [`docs/opencode-integration.md`](docs/opencode-integration.md)、[`docs/pi-integration.md`](docs/pi-integration.md)、[`docs/grok-integration.md`](docs/grok-integration.md)、[`docs/deepseek-integration.md`](docs/deepseek-integration.md) 与 [`docs/omp-integration.md`](docs/omp-integration.md) +- **自定义模型端点**(1.29.0 新增,目前仅 HTTP API)—— 让某个会话的 CLI 指向任意 OpenAI 兼容端点,而不是它自己的官方后端:本地的 llama.cpp、llama-swap、Ollama 或 vLLM 机器,也可以是 Azure AI Foundry、OpenRouter 这类云端网关。端点只需保存一次(`POST /api/model-endpoints`,模型列表从它的 `/v1/models` 自动发现),再应用到会话(`POST /api/sessions/:id/custom-model`),CLI 就会在原地重启并接上该端点。Claude、OpenCode、Pi、Grok 与 OMP 已实测通过;Codex、Gemini 与 DeepSeek 存在已记录的缺口,Antigravity 没有可用机制。工具栏选择器是下一步。详见 [`docs/custom-model-endpoints.md`](docs/custom-model-endpoints.md) +- **Web 标签页** —— 把 Grafana、Uptime Kuma、一个 Vite 开发服务器或任何仪表盘 URL 作为标签页打开在会话旁边(Run 下拉菜单 → **Web / URL** → **Add dashboard**)。仪表盘通过 Codeman 自己的源代理,因此 `http://` 目标在手机上走 HTTPS 也能用、走隧道也能用;单页应用能在自己的路径上正常路由,页面自己重载后也能自行恢复。智能体打印出的 `localhost` 链接会自动以 Web 标签页打开。详见 [`docs/web-tabs.md`](docs/web-tabs.md) +- **Docker 会话** —— 在隔离且加固的容器中运行 case。**Create New** 上勾选一个复选框即可用合理的默认值启动容器并在其中启动智能体;同一 case 的多个会话共享一个容器,也可以把 case 挂到你已经在跑的容器上;可将容器连同工作区导出为可移植的 `.tar.gz`,迁移到另一台机器。详见 [`docs/docker-cases.md`](docs/docker-cases.md) +- **远程 SSH 会话** —— 把 case 指向另一台机器,让智能体在那里一个持久的远程 tmux 中运行:SSH 断连不中断任务、自动重连,还能发现并附着主机上已在运行的会话;文件预览与下载走同一条 ssh 连接。详见 [`docs/remote-sessions.md`](docs/remote-sessions.md) - **Effort 与 Ultracode** —— 设置每会话的默认 effort(`low`–`max`),或启用 **ultracode**(动态多智能体工作流)。这些都只是软默认值 —— 会话中可随时用 `/effort` 切换。扩展思考预算也可配置 -- **语音输入** —— 用 Deepgram Nova-3 口述提示(带 Web Speech API 回退):切换录音、自动静音停止、实时音量表(`Ctrl+Shift+V`) +- **语音输入** —— 用 Deepgram Nova-3 口述提示,或者干脆用这台机器的 Claude Code 登录、不需要任何 API key(App Settings → Voice;带 Web Speech API 回退):切换录音、自动静音停止、实时音量表(`Ctrl+Shift+V`) - **图像输入** —— 直接把图片粘贴或拖放进会话 -- **手势控制** _(可选)_ —— 一个 MediaPipe 手部追踪叠加层,可徒手抓取/拖动会话窗口并捏合按钮。用 `CODEMAN_GESTURE=1` + App Settings → Display 启用 +- **手势控制** _(可选)_ —— 一个 MediaPipe 手部追踪叠加层,可徒手抓取/拖动会话窗口并捏合按钮。用 `CODEMAN_GESTURE=1` + App Settings → Terminal & Input 启用 - **多显示器横跨** _(macOS)_ —— 一键打开一个横跨所有显示器最大化的浏览器窗口,让浮动的智能体/手势面板可以跨越物理拼接缝 -- **文件查看器按钮** _(可选)_ —— 头部新增一个按钮,一键切换内置文件浏览器面板;在 App Settings → Display → Header Displays 中启用 -- **CJK / 输入法支持** —— 完整支持中文 / 日文 / 韩文的组合输入 +- **文件查看器按钮** _(可选)_ —— 页头新增一个按钮,一键切换内置文件浏览器面板;在 App Settings → Header & Panels → Header buttons 中启用 +- **CJK / 输入法支持** —— 完整支持中文 / 日文 / 韩文的组合输入,Ctrl、Alt 修饰的导航键也会原样透传给 CLI +- **页头里的套餐用量** —— 页头实时显示 Claude 订阅用量(5 小时窗口与每周窗口),数据来自 Codeman 在拉起 `claude` 时临时交给它的 statusline 导出器,绝不会写进你的设置文件;Codex 的限额则来自它自己的 app-server。按设备生效:桌面默认开,手机默认关 +- **会话列表,随你摆** —— 页头横条、带筛选框的左侧边栏,或一条竖向导轨,导轨的详细行带有创建时间与状态时长并按活动状态排序;手机首页和桌面首页导轨用的是同一套顺序 +- **终端外观** —— 七套皮肤(其中四套浅色)、按设备保存的字体与字重(内置的 JetBrains Mono 覆盖 100 到 800 的字重),以及可选启用的入场动画,覆盖标签、智能体窗口、终端面板和连接线 - **操作系统通知与主机名感知标题** —— 桌面提醒与标签标题以 `codeman:` 为前缀,使多主机配置不再含糊 --- @@ -416,7 +470,8 @@ PTY 输出 → 16ms 服务端批处理 → DEC 2026 包裹 → SSE → 客户端 - **资源模板** —— 展开复选框可选 **Small / Medium / Large / GPU** 预设(内存、CPU、GPU),也可以完全自定义。**磁盘是弹性的** —— 存储随数据增长,没有固定上限。 - **按案例共享容器** —— 多个会话可以 `docker exec` 进同一个容器;结束某个会话绝不会影响其他会话所在的容器。 - **默认加固** —— 非 root、`--cap-drop ALL`、`no-new-privileges`、PID/内存上限,绝不使用 `--privileged` 或 docker socket;**密封(sealed)** 配置(不注入主机凭据、关闭网络)只需一个开关。 -- **无感认证、凭据隔离** —— 主机上的 Claude / Codex / Antigravity / Gemini / OpenCode / Pi 登录在容器内开箱即用:凭据在启动时以只读种子方式复制注入,onboarding/信任提示已预先答复,不会弹出登录向导。容器保留自己的副本,绝不回写主机的凭据存储;跨边界共享的只有对话转录,导出文件也绝不包含机密。 +- **无感认证、凭据隔离** —— 主机上的 Claude / Codex / Antigravity / Gemini / OpenCode / Pi / Grok / OMP 登录在容器内开箱即用:凭据在启动时以只读种子方式复制注入,onboarding/信任提示已预先答复,不会弹出登录向导。容器保留自己的副本,绝不回写主机的凭据存储;跨边界共享的只有对话转录,导出文件也绝不包含机密。 +- **挂到你已经在跑的容器上** —— 在 Docker 面板勾选 **Attach to an existing container**,就能把 case 链接到一个现成容器,而不是新建一个。Codeman 只 `exec` 进去,绝不启动、停止、重启或删除它;一个被接管的容器可以在不同目录下支撑多个 case,**复制一个已有 case** 会用同一容器上的兄弟 case 预填表单。多用户模式下仅管理员可用,因为容器的挂载属于启动它的人。 - **迁移到另一台机器** —— 把容器的完整环境(工具链 + 工作区)导出为可移植的 `.tar.gz`,在另一台机器上导入到新案例即可继续。 - **持久耐用** —— Codeman 重启后重连会回到同一个存活的智能体;容器停止/重启后则从绑定挂载的转录恢复对话。 @@ -433,6 +488,7 @@ PTY 输出 → 16ms 服务端批处理 → DEC 2026 包裹 → SSE → 客户端 - **发现与附着**:列出主机上已在运行的 `codeman-*` 会话(由那台机器自己的 Codeman 或其他操作者启动)并附着其一。非你所有的已附着会话在关闭标签时**只分离,绝不杀掉**。 - **共享会话**:多个客户端可以以不同窗口尺寸同时附着同一个远程会话而互不挤压;发现列表会显示带客户端计数的「shared」徽标。 - **注入安全**:所有 ssh 命令行都经由单一的 shell 转义构建器生成,主机/路径/身份文件字段均有模式校验。 +- **文件也行**:远程 case 里的预览、下载和文本读取走同一条 ssh 连接(一次 `realpath` + `stat` 探测,然后流式 `cat`,支持 `Range` 拖动进度),所以点一个路径打开的就是智能体所在那台机器上的文件。什么都不会复制到 Codeman 主机;编辑和 Office 预览会明确返回 400,而不是一个误导性的 404。 在 **New Case → Remote** 中配置(主机、用户、身份文件、可选跳板机)。完整设计:[`docs/remote-sessions.md`](docs/remote-sessions.md)。 @@ -486,7 +542,7 @@ codeman users list systemctl --user enable codeman-tunnel loginctl enable-linger $USER -# 或通过 Codeman Web UI:Settings → Tunnel → 切换为开 +# 或通过 Codeman Web UI:App Settings → System → Remote access → Cloudflare Tunnel ```
@@ -588,7 +644,7 @@ Codeman 默认用 `--dangerously-skip-permissions` 启动会话,因此 Web UI - **默认仅环回** —— 绑定 `127.0.0.1`,仅可从本机访问,因此「无密码」默认配置开箱即安全。在未设置 `CODEMAN_PASSWORD` 的情况下绑定非环回主机会*启动但打印一条醒目警告*,并给出三个具体修复方案(设置密码、环回 + 一个带认证的隧道,或用 `--allow-unauthenticated-network` 显式确认) - **可选认证,真实会话** —— 通过 `CODEMAN_USERNAME`(默认 `admin`)/ `CODEMAN_PASSWORD` 的 HTTP Basic 认证。成功后签发一个不透明的 256 位 `codeman_session` cookie(`randomBytes(32)`)—— 服务端校验,而非客户端签名,因此无法离线伪造(24h TTL、自动延长、设备上下文审计日志) - **按 IP 速率限制** —— 失败 10 次 → `429` 并带 `Retry-After`(15 分钟衰减)。即便攻击者在同一 IP 上猛攻,有效 cookie 或正确密码也能*立即*恢复 —— 这很重要,因为所有隧道流量共享同一个环回 IP。二维码认证有自己独立的限制器 -- **可配置的权限模式**:`--dangerously-skip-permissions` 只是默认值。**App Settings → Claude CLI → Startup Mode** 可以把新会话切换为 Anthropic 的分类器护栏 `auto` 模式(低打扰,需要 Claude Code 2.1.207+)、`normal` 提示模式,或一份显式的允许工具列表。多用户模式下,未获授权的用户会被强制为 `auto`,shell 会话与跳过权限需要按用户显式授权 +- **可配置的权限模式**:`--dangerously-skip-permissions` 只是默认值。**App Settings → Agents & CLIs → Claude → Startup Mode** 可以把新会话切换为 Anthropic 的分类器护栏 `auto` 模式(低打扰,需要 Claude Code 2.1.207+)、`normal` 提示模式,或一份显式的允许工具列表。多用户模式下,未获授权的用户会被强制为 `auto`,shell 会话与跳过权限需要按用户显式授权 ### 始终开启的浏览器加固(v0.9.5) @@ -602,8 +658,8 @@ Codeman 默认用 `--dangerously-skip-permissions` 启动会话,因此 Web UI ### 输入、文件与响应头 -- **模式校验的输入** —— 每个 API 请求体都用 Zod v4 模式检查;一个 `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `ANTIGRAVITY_*` / `GEMINI_*` / `GOOGLE_*` / `PI_*` 环境变量前缀允许列表把控每个 CLI 能接收哪些设置 -- **路径限定** —— 文件路由在边界检查前先 `realpath`(无 TOCTOU);`..`、绝对路径、以及解析到工作目录之外的符号链接都会被拒绝。上限:10 MB 文本预览 / 50 MB 原始与下载;`/api/download` 对敏感路径(`.env`、`*credentials*`、`~/.ssh/`、`.aws/credentials`)做黑名单。SVG/HTML 以 `octet-stream` + `nosniff` + attachment 提供,因此会被下载而非执行 +- **模式校验的输入** —— 每个 API 请求体都用 Zod v4 模式检查;一个 `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `ANTIGRAVITY_*` / `GEMINI_*` / `GOOGLE_*` / `PI_*` / `GROK_*` / `XAI_*` / `DSH_*` / `DEEPSEEK_*` / `OMP_*` 环境变量前缀允许列表把控每个 CLI 能接收哪些设置,而那些能把 CLI 流量改道的键(base URL、配置目录)对非管理员用户会被钳制 +- **路径限定** —— 文件路由在边界检查前先 `realpath`(无 TOCTOU);`..`、绝对路径、以及解析到工作目录之外的符号链接都会被拒绝。上限:10 MB 文本预览 / 2 GB 原始与下载(`CODEMAN_MAX_DOWNLOAD_BYTES`;响应体是流式的并支持 `Range` 请求,所以这个上限只是合理性边界,不是内存保护);`/api/download` 对敏感路径(`.env`、`*credentials*`、`~/.ssh/`、`.aws/credentials`)做黑名单。SVG/HTML 以 `octet-stream` + `nosniff` + attachment 提供,因此会被下载而非执行 - **安全响应头** —— `Content-Security-Policy`(`default-src 'self'`,每个例外都逐条列举)、`X-Content-Type-Options: nosniff`、`X-Frame-Options: SAMEORIGIN`、HTTPS 下的 HSTS,以及**仅**对 `localhost` / `127.0.0.1` / `::1` 反射的 CORS ### 供应链与隔离 @@ -615,6 +671,22 @@ Codeman 默认用 `--dangerously-skip-permissions` 启动会话,因此 Web UI --- +## 终端界面(`codeman tui`) + +一个在终端里运行的全屏会话仪表盘。状态与 Web UI 完全一致,因为它就是同一个服务器的客户端: + +```bash +codeman tui # 仪表盘 +codeman tui --list # 带编号的会话列表,随即退出(可用于脚本) +codeman tui 2 # 直接附着到列表里的第 2 个会话 +``` + +会话按 **NEEDS YOU → WORKING → IDLE → RECENT** 分组,等得最久的排最前。`↑↓`/`j`/`k` 选择,`1`-`9` 与 `[`/`]` 切换会话,`Enter` 附着进 tmux 面板(按 **`F1`** 回来)。在面板里,顶部的横条会一直显示会话条,`Alt+1`-`Alt+9` 不用离开就能切换。`y`/`n`/数字可以直接在列表里回答待处理的权限对话框,`p` 发送一行提示,`n` 新建会话并直接进入,`x` 杀掉一个(`y` 确认),`/` 搜索,`g` 显示离开摘要,`?` 是帮助,`q` 退出。窄于 72 列时它会去掉预览面板、变成单列列表,所以在手机上的 Termius 里依然好用。没有服务器在跑时,它仍会以仅附着的降级模式启动。 + +Web UI 仍是主要界面;完整指南见 **[docs/tui.md](docs/tui.md)**(英文)。 + +--- + ## 键盘快捷键 > Ctrl 绑定在 macOS 上也接受 Cmd。 @@ -626,15 +698,21 @@ Codeman 默认用 `--dangerously-skip-permissions` 启动会话,因此 Web UI | `Ctrl/Cmd+Tab` | 下一个会话 | | `Alt/Option+[` / `Alt/Option+]` | 上一个 / 下一个会话 | | `Alt/Option+1`–`Alt/Option+9` | 切换到第 N 个标签(按物理键位,macOS Option 布局也适用) | +| `Alt/Option+B` | 折叠 / 展开会话侧边栏(仅侧边栏布局) | | `Ctrl+Shift+{` / `Ctrl+Shift+}` | 将当前标签左移 / 右移 | | `Ctrl/Cmd+C` | 复制选中内容;未选中时中断代理 | | `Ctrl+Shift+C` | 复制选中内容(永不中断) | +| `Ctrl/Cmd+V` | 粘贴,或上传剪贴板里的图片并粘贴其路径 | | `Ctrl/Cmd+L` | 清屏 | | `Ctrl+Shift+R` | 恢复终端尺寸 | | `Ctrl+Shift+V` | 切换语音输入 | | `Ctrl/Cmd +` / `-` | 字体大小 | | `Ctrl/Cmd+?` | 键盘帮助 | | `Shift+Enter` | 插入换行(发送到终端) | +| `Shift+拖动` | 在鼠标事件交给 CLI 的面板里选中文本 | +| 右键 | 复制选中内容(没有选中时保留原生菜单) | +| `Shift+滚轮` | 滚轮被转发给 CLI 时,滚动本地回滚缓冲区 | +| `Ctrl+Z` | 在智能体会话里被吞掉,运行中的 CLI 不会被挂起;shell 里照常是作业控制 | | `Escape` | 关闭面板与模态框 | --- @@ -643,16 +721,78 @@ Codeman 默认用 `--dangerously-skip-permissions` 启动会话,因此 Web UI 面向不经浏览器控制 Codeman 的 AI 智能体与自动化:一个拉起工作会话的智能体、一个 CI 机器人,或是**运行在 Codeman 会话*内部*、编排其他会话的 Claude Code**。UI 能做的一切都是 HTTP + CLI,因此智能体也能做。 -> **捷径:装上打包好的智能体技能。** 下面这一整套(外加多工作会话的实战配方)已经作为 Claude Code 技能随仓库发布在 [`skills/codeman`](skills/codeman/SKILL.md),会话内部的智能体不必等你把文档粘进提示词就能驱动 Codeman。三种获取方式: -> -> - `npx skills add Ark0N/Codeman --skill codeman -g`:全局安装,任何支持技能的智能体都能用 -> - Claude Code 插件:`/plugin marketplace add Ark0N/Codeman`,然后 `/plugin install codeman@codeman`:通过 Claude Code 自带的插件管理器全局安装,`/plugin update codeman` 跟随新版本;与 `codeman skill install` 二选一,两者都装会让技能出现两次(`codeman` 和 `codeman:codeman`) -> - `codeman skill install`(全局)或 `codeman skill install --case `:给那些从 npm 安装、从未克隆过仓库的用户;`codeman skill uninstall` 可撤销 -> - **App Settings → Agent Skill**(`agentSkillEnabled`,默认关闭):开启后,Codeman 会在每次于某个 case 中创建 Claude 会话时把技能注入该 case;case 里用户自己写的 `skills/codeman` 永远不会被覆盖 -> -> 全局安装(`codeman skill install` 或 `npx skills add`)会被**本机每一个新建的 Claude Code 会话**读到,无论它在不在 Codeman 里。技能自带门禁:不在 Codeman 会话中(`CODEMAN_MUX` 未设置)时它拒绝动作,所以全局装上它对无关会话没有代价。 -> -> ⚠️ 把 `agentSkillEnabled` 关回去**不会删掉已经注入的副本**(在创建时做清扫,会把技能从共用同一个 `.claude/` 目录的其他活动会话脚下抽走)。要删就按 case 删:`codeman skill uninstall --case `。 +### 智能体技能(从这里开始) + +这一节的所有内容也打包成了一个 **Claude Code 技能**,位于 [`skills/codeman`](skills/codeman/SKILL.md)。装一次,就再也不用把 API 文档粘进提示词。你用大白话说想要什么,已经坐在 Codeman 会话里的智能体会自己加载配方并驱动 API。 + +#### 第 1 步:安装 + +| 方式 | 命令 | 范围 | +| ---------------- | ---------------------------------------------------------- | ------------------------------------------------------------------------------------------ | +| Skills CLI | `npx skills add Ark0N/Codeman --skill codeman -g` | 全局,任何支持技能的智能体都能用 | +| Claude Code 插件 | `/plugin marketplace add Ark0N/Codeman`,然后 `/plugin install codeman@codeman` | 全局,通过 Claude Code 自带的插件管理器;`/plugin update codeman` 跟随新版本。与 `codeman skill install` 二选一:两者都装会让技能出现两次(`codeman` 和 `codeman:codeman`) | +| 内置 CLI | `codeman skill install` | 全局(`~/.claude/skills/codeman`),给那些从 npm 安装、从未克隆过仓库的用户 | +| 内置 CLI | `codeman skill install --case ` | 仅一个 case | +| Web UI | App Settings → Agents & CLIs → Claude → **Agent Skill** | 每次在某个 case 创建 Claude 会话时自动注入(`agentSkillEnabled`,跨设备同步,默认关闭) | + +`codeman skill uninstall [--case ]` 可以撤销 CLI 安装,并且绝不会碰你自己写的 `skills/codeman`。 + +#### 第 2 步:开口要 + +整个界面就这么多。不用 curl,不用端点名,不用会话 id。下面这些提示照原样就能用: + +| 你说 | 技能做的事 | +| ------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------- | +| _「现在有哪些会话在跑?」_ | 列出它们的名字、模式和状态。只读,随时可以问。 | +| _「在 `myapp` case 上起一个 shell 工作会话,跑测试套件,告诉我过没过。」_ | 拉起、等待一个拆开的完成标记、读回退出码、清理。 | +| _「起 3 个工作会话分别跑 lint、typecheck 和测试。并行跑,报告失败的。」_ | 扇出流程:每个任务一个会话,先全部启动,再逐个收集完成的。 | +| _「让一个 claude 工作会话在 `refactor-auth` 上总结 `src/session.ts`,然后关掉它。」_ | 拉起、走完就绪阶梯(包括首次运行的信任对话框)、发送并等待、读取干净的 transcript 答案、删除。 | +| _「盯着会话 w4,如果它卡在权限提示上就告诉我。」_ | 阻塞在 `blocked` 信号上,并把问题交给**你**。它绝不会替另一个会话回答提示。 | + +#### 第 3 步:没有了 + +智能体会删掉它启动的每一个会话。你可以在仪表盘里看着标签出现又消失。 + +#### 一次真实的运行,从头到尾 + +> **你:** 起 3 个 shell 工作会话,并行跑 lint / typecheck / 前端语法检查,告诉我哪个失败了。 + +```text +lint -> 9f2d8e5f dispatched +typecheck -> aff9c691 dispatched 仪表盘里出现 3 个标签 +syntax -> be9f1f15 dispatched + +lint DONE_lint_17909 rc=0 +typecheck DONE_typecheck_3409 rc=0 每完成一个就收集一个 +syntax DONE_syntax_18501 rc=0 + +deleted 9f2d8e5f, aff9c691, be9f1f15 标签消失 +``` + +那些 `DONE__` 字符串就是技能的**拆分标记**技巧,也是扇出在没有 hook 的 `shell` 会话上依然可靠的原因:敲进去的那一行只含 `${M}_17909`,因此只有命令真正的*输出*里才会出现 `DONE_17909`。不拆开的标记会在命令还没跑之前就匹配到你自己按键的回显。 + +#### 盒子里有什么 + +| 文件 | 内容 | +| ------------------------------------------------------------------- | ------------------------------------------------------------------------------------------- | +| [`SKILL.md`](skills/codeman/SKILL.md) | 安全规则、现成的快速路径(起 N 个工作会话、派任务、收集)和动词索引。始终加载。 | +| [`reference/verbs.md`](skills/codeman/reference/verbs.md) | 14 个动词的详细说明:就绪、发送并等待、标记、中断、清理。按需加载。 | +| [`reference/recipes.md`](skills/codeman/reference/recipes.md) | 8 个完整流程:claude、DeepSeek Harness 与 shell 工作会话、扇出、盯住被卡住的工作会话、消息扇出。按需加载。 | +| [`reference/endpoints.md`](skills/codeman/reference/endpoints.md) | 完整端点表、错误码、各模式的信号表、容量限制。按需加载。 | +| [`reference/messaging.md`](skills/codeman/reference/messaging.md) | 通过 Claude Code 跨会话消息直接和 claude 工作会话对话。按需加载。 | + +里面的每一个配方都在真实服务器上验证过,注释记录的是实测出来而不是猜出来的失败模式。 + +#### 两件值得知道的事 + +- **它会自我门禁。** 不在 Codeman 会话里(`CODEMAN_MUX` 未设置)时,技能拒绝动作,也不去猜 API 地址,所以全局安装对无关的 Claude Code 会话没有任何代价。 +- **它刻意保守。** 未经提示,它只会拉起会话、给它们发提示,并删除**它在同一段对话里自己创建的**会话(按精确 id,经由一个拒绝删除智能体自身会话的失败即关闭守卫)。删除 case(会抹掉一个真实的代码目录)、批量杀会话、改动 respawn/ralph/cron/orchestrator 以及写设置,都需要你开口并指名目标。 + +⚠️ 把 `agentSkillEnabled` 关回去**不会删掉已经注入的副本**(在创建时做清扫,会把技能从共用同一个 `.claude/` 目录的其他活动会话脚下抽走)。要删就按 case 删:`codeman skill uninstall --case `。 + +--- + +**这一节余下的部分是手动路径**:同样的操作用裸 HTTP 来做,适合 CI 机器人、shell 脚本,或任何不支持技能的智能体。 ### 检测自己身处 Codeman 内部 @@ -673,7 +813,7 @@ Codeman 默认用 `--dangerously-skip-permissions` 启动会话,因此 Web UI 4. **响应信封。** 多数端点返回 `{ "success": true, "data": … }`(错误:`{ "success": false, "error", "errorCode" }`)。少数遗留 GET 返回裸响应体 —— **两种都要处理**(`body.data ?? body`)。 5. **`/api/v1/*`** 是 `/api/*` 的稳定别名。 6. **用等待代替轮询,别把超时当成错误。** 等待类端点在没等到事情发生时也以 HTTP `200` 加 `wait.timedOut: true` 应答,所以要循环调用短等待(默认 60 秒),而不是发一个超长的调用:隧道会掐断空闲连接。`wait.timeoutMs` 告诉你服务端钳制之后真正采用的超时(上限 600 秒)。 -7. **只有 `claude` 会话会发出 `stop` 与 `blocked`。** 这两个来自 Claude Code hook;`shell` 与外部 CLI(opencode/codex/gemini/antigravity/pi)只接受 `idle`、`working` 与 `exit`。在这些模式上显式索要 `stop` 会得到 `400`;不传 `until` 则永远安全。⚠️ `shell` 会话的 `idle` 只在启动时触发**一次**,此后再也不会,所以在那里用「发送并等待」只能等到超时:没有 hook 的会话请用 `wait-output` 标记来同步。 +7. **只有 `claude` 与 `deepseek` 会话会发出 `stop` 与 `blocked`。** 这两个来自 hook(Claude Code 自己的,以及 DeepSeek Harness 的状态桥接);`shell` 与其他外部 CLI(opencode/codex/gemini/antigravity/pi/grok/omp)只接受 `idle`、`working` 与 `exit`。在这些模式上显式索要 `stop` 会得到 `400`;不传 `until` 则永远安全。⚠️ `shell` 会话的 `idle` 只在启动时触发**一次**,此后再也不会,所以在那里用「发送并等待」只能等到超时:没有 hook 的会话请用 `wait-output` 标记来同步。 8. **没有任何东西会报告「就绪」,得自己显式等。** 新会话在 PID 出现之前一律回答 `{"signal":"exit","immediate":true}`(意思是*还没启动*,不是*崩了*),而全新 case 里的 `claude` 工作会话接着会停在 CLI 的信任对话框上。此时给它发提示,等待会在约 2 秒后因 `idle` 解除,看上去和一个跑完的回合一模一样,而文本其实卡在对话框里。下面的配方 2b 就是避开它的顺序。 ### 常用配方 @@ -738,7 +878,7 @@ curl -sG "$API/api/sessions/$SID/wait-output" \ --data-urlencode "match=DONE_$N" --data-urlencode 'from=buffer' \ --data-urlencode 'timeout=60000' | jq '.data.wait' -# 5. 读回答案。claude / codex 会话用 last-response:它取自 transcript 而不是屏幕, +# 5. 读回答案。claude / codex / deepseek 会话用 last-response:它取自 transcript 而不是屏幕, # 因此不带 TUI 的画框与重画噪声。⚠️ 要轮询,别只读一次:transcript 落盘比 stop # 信号稍晚,紧跟着「发送并等待」返回后立刻读,常常拿到空串。 for _ in $(seq 1 10); do @@ -747,7 +887,7 @@ for _ in $(seq 1 10); do done printf '%s\n' "$TXT" -# 5b. 其他模式(shell/opencode/gemini/antigravity/pi)没有 transcript,读终端。 +# 5b. 其他模式(shell/opencode/gemini/antigravity/pi/grok/omp)没有 transcript,读终端。 # ⚠️ 用 terminal?tail=,不要用 /output:后者的 textOutput 对每个由 tmux 承载的 # (也就是每个交互式)会话都是空的。tail 按字节计,返回的是含 ANSI 的终端数据。 curl -s "$API/api/sessions/$SID/terminal?tail=8000" | jq -r '.data.terminalBuffer' @@ -780,7 +920,9 @@ codeman session start -d /path/to/repo # (s) 启动会话 codeman session list # 列出会话 codeman session logs # 查看输出 codeman task add "fix the failing test" # (t) 排入任务 -codeman attach # 附着 Claude hook 上下文 +codeman attach # 为本地文件显示一张附件卡片 +codeman tui --list # 带编号的会话列表(管道输出时为纯文本) +codeman tui 3 # 附着到该列表里的第 3 个会话 ``` ### Hook(事件*回流*到 Codeman) @@ -793,7 +935,7 @@ Codeman 会注册 Claude Code hook,它们 `POST /api/hook-event`(`permission ## API -基于 Fastify 的 REST —— **21 个路由模块中约 200 个处理器**,外加一条 SSE 流和一条 WebSocket 终端通道。所有响应都使用 `ApiResponse` 信封(`{success, data}` / `{success, error, errorCode}`);`/api/v1/*` 是稳定别名。以下是一个有代表性的子集: +基于 Fastify 的 REST —— **25 个路由模块中约 230 个处理器**,外加一条 SSE 流和一条 WebSocket 终端通道。所有响应都使用 `ApiResponse` 信封(`{success, data}` / `{success, error, errorCode}`);`/api/v1/*` 是稳定别名。以下是一个有代表性的子集: ### 会话(Sessions) @@ -804,11 +946,13 @@ Codeman 会注册 Claude Code hook,它们 `POST /api/hook-event`(`permission | `POST` | `/api/sessions/:id/input` | 发送输入(`{input, useMux?, clientId?, seq?, wait?, waitTimeout?}`:`clientId`+`seq` = 精确一次;`wait` 阻塞到这一回合结束) | | `GET` | `/api/sessions/:id/terminal` | 读取终端输出(`?tail=`、`?full=1`):交互式会话的读取路径 | | `GET` | `/api/sessions/:id/output` | 一次性的解析输出(tmux 承载的会话里 `textOutput` 为空) | +| `GET` | `/api/sessions/:id/last-response` | 从 transcript 读出的最后一条回答,纯文本(claude、codex、deepseek) | | `GET` | `/api/sessions/:id/wait` | 阻塞到某个信号触发(`?until=stop,idle,exit&timeout=&fresh=`);超时是 `200` | | `GET` | `/api/sessions/:id/wait-output` | 阻塞到某个字面串出现(`?match=&nocase=&from=now\|buffer&timeout=`) | | `GET` | `/api/sessions/unified` | 统一的活动 + 历史清单(会话管理器):`?q=&limit=` | | `POST` | `/api/sessions/:id/pin` | 在会话管理器中置顶 / 取消置顶(`{pinned}`) | | `PUT` | `/api/session-order` | 跨设备同步标签顺序(`{order: [ids]}`) | +| `POST` | `/api/sessions/:id/custom-model` | 让会话的 CLI 在一个已保存的自定义端点上原地重启(`{endpointId, modelId}`;`{clear: true}` 回到官方后端) | | `DELETE` | `/api/sessions/:id` | 删除会话 | ### 重生(Respawn) @@ -857,6 +1001,7 @@ Codeman 会注册 Claude Code hook,它们 `POST /api/hook-event`(`permission | `GET` | `/api/system/update/check` | 检查新发行版 | | `POST` | `/api/system/update` | 自更新(git-clone 安装) | | `POST` | `/api/clipboard` | 把文本推送到所有已连接浏览器(`{text}`) | +| `GET` / `POST` | `/api/model-endpoints` | 列出 / 保存自定义的 OpenAI 兼容端点(`PUT` / `DELETE` `/:id`;多用户模式下仅管理员) | | `GET` | `/api/sessions/:id/run-summary` | 时间线 + 统计 | > **想在 Codeman 之上做集成?**[`docs/extending-codeman.md`](docs/extending-codeman.md)(英文)是集成指南:把你自己的界面作为标签页嵌入、订阅 SSE 事件流以便在 agent 需要你时做出响应、用脚本驱动 Codeman,以及动手前值得先了解的那些坑。Codeman 刻意不提供插件运行时,所以一个集成就是你自己的进程在讲 HTTP。 @@ -893,7 +1038,7 @@ flowchart TB end subgraph External["外部"] - CLI["AI CLI
Claude Code / OpenCode / Codex / Antigravity / Gemini / Pi"] + CLI["AI CLI
Claude Code / OpenCode / Codex / Antigravity / Gemini / Pi / Grok / DeepSeek / OMP"] BG["后台智能体
(Task 工具)"] end end @@ -931,6 +1076,12 @@ npm test # 运行测试(与 CI 相同;浏览器/移动端 --- +## 社区 + +提问、安装求助和想法都在 [GitHub Discussions](https://github.com/Ark0N/Codeman/discussions):[Q&A 板块](https://github.com/Ark0N/Codeman/discussions/categories/q-a)回答了最常见的那些(手机访问、通宵运行、更新),路线图则在 [Ideas](https://github.com/Ark0N/Codeman/discussions/categories/ideas) 里决定。Bug 请提到 [issues](https://github.com/Ark0N/Codeman/issues);报告通常一天内会得到回复,每个发行版都会点名感谢报告者和贡献者。想参与贡献?[CONTRIBUTING.md](.github/CONTRIBUTING.md) 是地图:皮肤、翻译和文档都是很好的第一个 PR,更大的特性先从一个 Discussion 开始。如果你对自己的配置很自豪,发到 [Show and tell](https://github.com/Ark0N/Codeman/discussions/300) 来。 + +--- + ## 代码库质量 本代码库经历了一次全面的 7 阶段重构,消除了上帝对象、集中了配置,并建立了模块化架构: @@ -954,7 +1105,7 @@ npm test # 运行测试(与 CI 相同;浏览器/移动端 [![npm](https://img.shields.io/npm/v/xterm-zerolag-input?style=flat-square&color=22c55e)](https://www.npmjs.com/package/xterm-zerolag-input) -为 xterm.js 提供即时按键反馈的叠加层。通过把输入的字符立即渲染为像素级精准的 DOM 叠加层,消除高 RTT 连接下的感知输入延迟。零依赖、可配置的提示符检测、带 78 个测试的完整状态机。 +为 xterm.js 提供即时按键反馈的叠加层。通过把输入的字符立即渲染为像素级精准的 DOM 叠加层,消除高 RTT 连接下的感知输入延迟。零依赖、gzip 后 6.1 kB、可配置的提示符检测、CJK/emoji 宽字符支持、带 238 个测试的完整状态机。 ```bash npm install xterm-zerolag-input @@ -977,3 +1128,8 @@ MIT —— 见 [LICENSE](LICENSE)

跟踪会话。可视化智能体。掌控重生。让它在你睡觉时持续运行。

+ +

+ 如果 Codeman 帮你省了时间,点个 star 能让更多人找到它。
+ 欢迎到 Issues 报告 bug 和提出特性想法。 +

From 3f2928ae730a647e89cef7c8817706fcbd65f0d2 Mon Sep 17 00:00:00 2001 From: Devvyn <22340871+opticon454@users.noreply.github.com> Date: Tue, 15 Sep 2026 09:10:37 +0800 Subject: [PATCH 04/28] chore(cli-registry): clean up dead code and stale claims left after #380 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses the "left as they are"/"worth knowing" items Ark0N named when merging #380 (the CLI-catalogue-driven install.sh + Docker agent image PR), none of which were correctness-blocking but all of which were real: - Removed install.sh's dead _cli_index/check_cli/get_cli_path helpers: the catalogue-driven menu and hints stopped calling them and nothing else ever did. - The generator no longer emits CLI_KIND/CLI_NPM, two bash arrays install.sh never read (the .mjs/docker-hosts.ts producers already read the JSON catalogue's kind/npmPackage fields directly, so only the bash copies were dead). - detect_all_clis now skips a disabled entry's probe entirely instead of running it and filtering the result downstream. No stock entry ships disabled today, so this closes a latent inefficiency before it is a latent bug rather than fixing an observed one. - The install hint for a launcherProfile entry (DeepSeek today) now explains in one line why it's a docs link and not a command: its own docs page documents `npm install -g @deepseek-ai/dsh`, which installs the launcher only and can't drive a pane, the exact trap the menu already avoids by withholding the command. Driven by a new generated CLI_LAUNCHER_ONLY array (from discovery.launcherProfile), not an id check, so any future launcherProfile entry gets the same caveat free. - Corrected the non-interactive-default comment: on a wget-only host, Claude's curl one-liner is filtered out of the offered list first, so the default becomes whichever npm-based entry sorts earliest instead (Codex today), not always Claude. Behaviour is unchanged — it was already printed, never silent — only the comment overclaimed. Tests: extended test/install-sh-invariants.test.ts with a positive guard for the new array and the trimmed array list, a negative guard that CLI_KIND/CLI_NPM/the three dead helpers cannot come back, and two real-bash tests (driven the same way the existing skip-menu tests are) proving a disabled entry is genuinely never probed rather than merely filtered after the fact. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_011WzDjJnbK7zug8iQWnCc9z --- .changeset/cli-catalog-followups.md | 24 +++++++++++ install.sh | 65 +++++++++++++---------------- scripts/generate-cli-catalog.mts | 13 +++--- test/install-sh-invariants.test.ts | 65 +++++++++++++++++++++++++++-- 4 files changed, 121 insertions(+), 46 deletions(-) create mode 100644 .changeset/cli-catalog-followups.md diff --git a/.changeset/cli-catalog-followups.md b/.changeset/cli-catalog-followups.md new file mode 100644 index 00000000..5970e119 --- /dev/null +++ b/.changeset/cli-catalog-followups.md @@ -0,0 +1,24 @@ +--- +"aicodeman": patch +--- + +Cleans up the loose ends the maintainer flagged as "worth knowing rather than fixing" when +merging the CLI-catalogue-driven `install.sh`/Docker-agent-image PR (#380): + +- `install.sh` no longer carries `_cli_index`/`check_cli`/`get_cli_path`, three generic + lookup helpers left behind once the catalogue-driven menu and hints stopped calling them. +- The generator no longer emits `CLI_KIND`/`CLI_NPM`, two bash arrays nothing in `install.sh` + read (the `.mjs`/`docker-hosts.ts` producers already read the JSON catalogue's `kind`/ + `npmPackage` fields directly). +- `detect_all_clis` now skips a disabled entry entirely rather than probing it and filtering + the result downstream — no stock entry ships disabled today, so this is a latent + inefficiency closed before it is a latent bug, not a behaviour change. +- The install hint for a `launcherProfile` entry (DeepSeek today) now explains, in one line, + why it is a docs link rather than a runnable command — its docs page documents + `npm install -g @deepseek-ai/dsh`, which installs the launcher only and cannot drive a pane + on its own, the exact trap the menu already avoids by withholding the command. Driven by a + new generated `CLI_LAUNCHER_ONLY` array (from `discovery.launcherProfile`), not an id check. +- The non-interactive default's comment no longer claims it is always Claude Code: on a + wget-only host, Claude's curl one-liner is filtered out of the offered list first, so the + default becomes whichever npm-based entry sorts earliest instead. Behaviour is unchanged + (and was already printed, so never silent) — only the comment was wrong. diff --git a/install.sh b/install.sh index 5be359c2..8b5672bc 100755 --- a/install.sh +++ b/install.sh @@ -93,8 +93,7 @@ export PUPPETEER_SKIP_DOWNLOAD="${PUPPETEER_SKIP_DOWNLOAD:-1}" CLI_IDS=('claude' 'shell' 'opencode' 'codex' 'gemini' 'antigravity' 'pi' 'grok' 'deepseek' 'omp') CLI_LABELS=('Claude' 'Shell' 'OpenCode' 'Codex' 'Gemini' 'Antigravity' 'Pi' 'Grok' 'DeepSeek' 'OMP') CLI_ENABLED=(1 1 1 1 1 1 1 1 1 1) -CLI_KIND=('agent' 'shell' 'agent' 'agent' 'agent' 'agent' 'agent' 'agent' 'agent' 'agent') -CLI_NPM=('@anthropic-ai/claude-code' '' 'opencode-ai' '@openai/codex' '@google/gemini-cli' '' '@earendil-works/pi-coding-agent' '' '@deepseek-ai/dsh' '') +CLI_LAUNCHER_ONLY=(0 0 0 0 0 0 0 0 1 0) CLI_DOCS=('https://docs.claude.com/claude-code' '' 'https://opencode.ai/docs' 'https://developers.openai.com/codex/cli' 'https://github.com/google-gemini/gemini-cli' 'https://antigravity.google/cli' 'https://pi.dev' 'https://github.com/xai-org/grok-build' 'https://github.com/deepseek-ai/deepseek-harness' 'https://omp.sh') CLI_CMD_LINUX=('curl -fsSL https://claude.ai/install.sh | bash' '' 'curl -fsSL https://opencode.ai/install | bash' 'npm install -g @openai/codex' 'npm install -g @google/gemini-cli' 'curl -fsSL https://antigravity.google/cli/install.sh | bash' 'npm install -g --ignore-scripts @earendil-works/pi-coding-agent' 'curl -fsSL https://x.ai/cli/install.sh | bash' '' 'curl -fsSL https://omp.sh/install | sh') CLI_CMD_DARWIN=('curl -fsSL https://claude.ai/install.sh | bash' '' 'curl -fsSL https://opencode.ai/install | bash' 'npm install -g @openai/codex' 'npm install -g @google/gemini-cli' 'curl -fsSL https://antigravity.google/cli/install.sh | bash' 'npm install -g --ignore-scripts @earendil-works/pi-coding-agent' 'curl -fsSL https://x.ai/cli/install.sh | bash' '' 'brew install can1357/tap/omp') @@ -410,22 +409,6 @@ check_build_tools() { # test/install-sh-detection-parity.test.ts: the process PATH first (each declared # binary name in turn), then each known install path, dir-major. -# Index of "$1" in CLI_IDS -> CLI_IDX, returning 1 with CLI_IDX=-1 when unknown. -# A global rather than an echo because this runs inside loops, and a subshell per -# lookup is a fork per CLI per call site. -CLI_IDX=-1 -_cli_index() { - local want="$1" i - CLI_IDX=-1 - for ((i = 0; i < ${#CLI_IDS[@]}; i++)); do - if [[ "${CLI_IDS[$i]}" == "$want" ]]; then - CLI_IDX=$i - return 0 - fi - done - return 1 -} - # `dsh` is the hardest name of the lot: Debian ships an unrelated `dsh` # (dancer's shell). The server-side resolver settles it by demanding the # harness's own help banner; detection here only feeds the "you have no AI CLI" @@ -465,7 +448,8 @@ _cli_candidate_ok() { # Resolve every CLI in ONE pass, memoized. # -# CLI_FOUND_PATH is parallel to CLI_IDS ('' when not found). CLI_FOUND_COUNT +# CLI_FOUND_PATH is parallel to CLI_IDS ('' when not found, and also '' for a +# DISABLED entry — it is never probed at all, see below). CLI_FOUND_COUNT # counts only ENABLED entries that have a binary to look for, which is what the # "no AI CLI found" gate asks about — `shell` has no binary and must never make # that gate think an agent is installed. @@ -485,6 +469,16 @@ detect_all_clis() { for ((i = 0; i < ${#CLI_IDS[@]}; i++)); do found="" + # A disabled entry is never even probed: every consumer already filters + # on CLI_ENABLED before showing anything, so the command-v/stat calls + # below would be pure waste — and, unlike filtering downstream, skipping + # the probe here is what makes CLI_ENABLED mean "look for it" rather + # than just "offer it once found". + if [[ "${CLI_ENABLED[$i]}" != "1" ]]; then + CLI_FOUND_PATH[$i]="" + continue + fi + # 1. The process PATH, each declared binary name in turn. bin_end=$((${CLI_BIN_OFF[$i]} + ${CLI_BIN_LEN[$i]})) for ((j = ${CLI_BIN_OFF[$i]}; j < bin_end; j++)); do @@ -520,20 +514,6 @@ detect_all_clis() { return 0 } -# Is this CLI installed? Unknown id is "no", never an error. -check_cli() { - detect_all_clis - _cli_index "$1" || return 1 - [[ -n "${CLI_FOUND_PATH[$CLI_IDX]}" ]] -} - -# Where it was found, or nothing. -get_cli_path() { - detect_all_clis - _cli_index "$1" || return 1 - printf '%s\n' "${CLI_FOUND_PATH[$CLI_IDX]}" -} - # ---------------------------------------------------------------------------- # Catalogue helpers # ---------------------------------------------------------------------------- @@ -589,7 +569,12 @@ cli_catalog_names() { # the registry but an empty one here: installing the launcher alone leaves # nothing that can drive a pane, so the generator withholds the command for # any launcherProfile entry (see installCommandFor in generate-cli-catalog.mts) -# and this hint falls through to the docs URL instead. +# and this hint falls through to the docs URL instead — CLI_LAUNCHER_ONLY adds +# one line explaining WHY it is a docs link and not a command, so a user who +# follows that link straight to `npm install -g @deepseek-ai/dsh` (which the +# docs page itself documents) does not land back in the same "installed but +# cannot drive a pane" trap the menu exists to avoid. Data-driven, not an id +# check: any future launcherProfile entry gets the same caveat for free. cli_catalog_print_install_hints() { detect_all_clis local i @@ -601,6 +586,9 @@ cli_catalog_print_install_hints() { echo -e " ${CYAN}${CLI_INSTALL_CMD_TRUSTED[$i]}${NC} # ${CLI_LABELS[$i]}" elif [[ -n "${CLI_DOCS[$i]}" ]]; then echo -e " ${CLI_LABELS[$i]}: see ${CYAN}${CLI_DOCS[$i]}${NC}" + if [[ "${CLI_LAUNCHER_ONLY[$i]}" == "1" ]]; then + echo -e " (its package installs a launcher only — it needs a profile that can drive a pane, see the docs above)" + fi fi done } @@ -679,9 +667,12 @@ offer_ai_cli_install() { local cli_choice="" if [[ "$NONINTERACTIVE" == "1" ]] || ! has_tty; then - # Explicit automation opt-in: default to the first offered entry, - # which is registry order, which is Claude Code (order 0) — the - # same default this prompt has always taken non-interactively. + # Explicit automation opt-in: default to the first OFFERED entry. + # That is registry order, which is Claude Code (order 0), UNLESS + # this is a wget-only host and Claude's curl one-liner was just + # filtered out of offer_idx above — there, the first survivor is + # whichever npm-based entry sorts earliest (Codex today), not + # Claude. Printed either way so the choice is never silent. cli_choice="1" info "CODEMAN_NONINTERACTIVE=1: defaulting to ${CLI_LABELS[${offer_idx[0]}]}" else diff --git a/scripts/generate-cli-catalog.mts b/scripts/generate-cli-catalog.mts index ce9cd7e4..e17df0ec 100644 --- a/scripts/generate-cli-catalog.mts +++ b/scripts/generate-cli-catalog.mts @@ -135,8 +135,7 @@ export function renderInstallShBlock(entries: CliEntry[] = STOCK_CLIS): string { const ids: string[] = []; const labels: string[] = []; const enabled: string[] = []; - const kinds: string[] = []; - const npm: string[] = []; + const launcherOnly: string[] = []; const docs: string[] = []; const cmdLinux: string[] = []; const cmdDarwin: string[] = []; @@ -151,8 +150,11 @@ export function renderInstallShBlock(entries: CliEntry[] = STOCK_CLIS): string { ids.push(shQuote(entry.id as string)); labels.push(shQuote(entry.label)); enabled.push(entry.enabled ? '1' : '0'); - kinds.push(shQuote(entry.kind)); - npm.push(shQuote(entry.discovery.install.npmPackage ?? '')); + // Parallel to CLI_IDS: 1 when this entry's install command installs a launcher rather + // than something that can drive a pane on its own (see installCommandFor below). Purely + // derived from discovery.launcherProfile — install.sh's hint printer reads this to add a + // caveat instead of hardcoding which id it means. + launcherOnly.push(entry.discovery.launcherProfile ? '1' : '0'); docs.push(shQuote(entry.discovery.install.docsUrl ?? '')); cmdLinux.push(shQuote(installCommandFor(entry, 'linux'))); cmdDarwin.push(shQuote(installCommandFor(entry, 'darwin'))); @@ -195,8 +197,7 @@ export function renderInstallShBlock(entries: CliEntry[] = STOCK_CLIS): string { arr('CLI_IDS', ids), arr('CLI_LABELS', labels), arr('CLI_ENABLED', enabled), - arr('CLI_KIND', kinds), - arr('CLI_NPM', npm), + arr('CLI_LAUNCHER_ONLY', launcherOnly), arr('CLI_DOCS', docs), arr('CLI_CMD_LINUX', cmdLinux), arr('CLI_CMD_DARWIN', cmdDarwin), diff --git a/test/install-sh-invariants.test.ts b/test/install-sh-invariants.test.ts index 126c406d..abab770f 100644 --- a/test/install-sh-invariants.test.ts +++ b/test/install-sh-invariants.test.ts @@ -48,13 +48,12 @@ describe('install.sh generated-catalogue block', () => { ); }); - it('declares every array the detection code indexes', () => { + it('declares every array install.sh actually reads', () => { for (const name of [ 'CLI_IDS', 'CLI_LABELS', 'CLI_ENABLED', - 'CLI_KIND', - 'CLI_NPM', + 'CLI_LAUNCHER_ONLY', 'CLI_DOCS', 'CLI_CMD_LINUX', 'CLI_CMD_DARWIN', @@ -69,6 +68,17 @@ describe('install.sh generated-catalogue block', () => { } }); + it('declares no array install.sh never reads', () => { + // CLI_KIND and CLI_NPM were generated and read by nothing (the .mjs/docker-hosts.ts + // producers read the JSON's `kind`/`npmPackage` fields directly; only these two bash + // arrays were dead). A generated-but-unread array is a maintenance trap the generator + // itself cannot warn about — it has no reader to check against — so this pins the + // opposite of the test above: naming what must NOT come back rather than what must. + for (const name of ['CLI_KIND', 'CLI_NPM']) { + expect(new RegExp(`^${name}=\\(`, 'm').test(SOURCE), `${name} is declared but nothing reads it`).toBe(false); + } + }); + it('keeps no hand-written per-CLI detection behind', () => { // The nine `*_SEARCH_PATHS` arrays and eighteen `check_`/`get__path` pairs are // what this change removes. One left behind would be a second source of truth that the @@ -93,6 +103,16 @@ describe('install.sh generated-catalogue block', () => { [] ); }); + + it('keeps no dead generic-lookup helpers behind', () => { + // _cli_index/check_cli/get_cli_path were the ungenericized precursor to the per-CLI + // helpers above: same shape, one level of indirection, called from nowhere once the + // catalogue-driven menu and hints stopped needing a lookup-by-id. Unlike the per-CLI + // pairs these are exact names, not derived from the catalogue. + for (const fn of ['_cli_index()', 'check_cli()', 'get_cli_path()']) { + expect(CODE.includes(fn), `${fn} should have been removed as dead code`).toBe(false); + } + }); }); describe('install.sh trust boundary', () => { @@ -245,3 +265,42 @@ describe('install.sh AI CLI install menu', () => { expect(run.status).toBe(1); }); }); + +describe('install.sh detect_all_clis and a disabled entry', () => { + // No stock entry ships disabled today, so this is characterization rather than a regression + // pin on real data: it drives the real function in a real bash with entry 0 fabricated + // disabled, and points its binary at `bash` — guaranteed resolvable via `command -v` — to + // prove the entry is genuinely never PROBED (CLI_FOUND_PATH stays empty) rather than merely + // filtered out downstream by every consumer's own `CLI_ENABLED` check. + function driveDetect(disableEntry0: boolean) { + const driver = ` + set -euo pipefail + export CODEMAN_INSTALL_SH_LIB=1 + . "$1" + k=0; while [[ $k -lt \${#CLI_ALL_BINS[@]} ]]; do CLI_ALL_BINS[$k]="codeman-test-no-such-bin-$k"; k=$((k + 1)); done + k=0; while [[ $k -lt \${#CLI_ALL_PATHS[@]} ]]; do CLI_ALL_PATHS[$k]="/nonexistent/codeman-test/$k"; k=$((k + 1)); done + # Point entry 0's first declared binary at something that WILL resolve, so a probe that + # runs at all finds it. + CLI_ALL_BINS[\${CLI_BIN_OFF[0]}]="bash" + ${disableEntry0 ? 'CLI_ENABLED[0]="0"' : ''} + CLI_DETECT_DONE="" + detect_all_clis + echo "path0=[\${CLI_FOUND_PATH[0]}]" + echo "found=$CLI_FOUND_COUNT" + `; + const result = spawnSync('bash', ['-c', driver, 'bash', INSTALL_SH], { encoding: 'utf-8', timeout: 30_000 }); + return { status: result.status, stdout: result.stdout ?? '', stderr: result.stderr ?? '' }; + } + + it('probes an enabled entry (control case)', () => { + const run = driveDetect(false); + expect(run.stdout, run.stderr).not.toContain('path0=[]'); + expect(run.stdout).toContain('found=1'); + }); + + it('never probes a disabled entry', () => { + const run = driveDetect(true); + expect(run.stdout, run.stderr).toContain('path0=[]'); + expect(run.stdout).toContain('found=0'); + }); +}); From 5b920cb43d4eb36e3296ca8c718a55de48b95792 Mon Sep 17 00:00:00 2001 From: Codeman maintainer Date: Tue, 15 Sep 2026 17:59:16 +0200 Subject: [PATCH 05/28] feat(sessions): land auto-naming opt-in, in the prefix form, from the first user prompt only Finishes #376. The contributed keystroke tracker sat on the raw byte stream and named tabs wrong five ways (every prompt, every write path, a bare Esc eating the next prompt's first character, pasted newlines as Enter, any CSI clearing the draft) and replaced the whole name, which dropped the case from the tab and reset the w counter. This lands the feature with each of those closed: - First prompt means the first: applyAutoName() flips a placeholder to `auto` whether or not the string changed. nameSource is now the tri-state placeholder | auto | manual; the name setter is the only manual path. - Only user-originated input counts: write()/writeViaMux() take SessionWriteOptions.fromUser, set by the browser WS path and POST /input only, so Ralph, respawn, cron, approvals and the trust-dialog keys can never name a tab. A startMode 'shell' CLI never feeds the tracker (a capability, not an id check); the send-key route feeds trackUserInput() because its line feed bypasses the session. - Prefix form `w3-case: title`: parseSessionPrefix() already renders it as the title with the prefix in the tooltip and the next-session counter still matches it. Composed within MAX_SESSION_NAME_LENGTH. - Tracker rules per key: bare Esc resolves at chunk end; mouse/focus reports, Tab, cursor keys, Shift+Tab are no-ops; Up/Down and Ctrl+P/N/R taint the draft so Enter submits nothing rather than a fragment; bracketed-paste newlines and Ctrl+J / Shift+Enter join with one space; the draft keeps its head past 8192 code points; an escape past 64 bytes is abandoned. - Title: slash commands by shape (a path is a prompt), `!` escapes refused, first sentence only past 8 code points ("e.g." is not a title), 72 code points on a word boundary. - Synced `autoNameSessions` setting, default OFF (the prompt reaches mux-sessions.json, session:updated and /api/search), App Settings -> Appearance -> Tabs, read fresh per prompt after the eligibility check. Tests: test/session-auto-name.test.ts (tracker, title, composition, ownership, emit gating), the wiring test (once, prefix, setting off, manual protected), test/routes/session-name-routes.test.ts (PUT /name flips to manual and persists). Verified live on an isolated instance: API and browser-typed prompts name the tab, a second prompt does not, shells and renamed tabs are untouched, nameSource survives a restart. Co-Authored-By: Claude Fable 5.1 --- .changeset/auto-name-sessions.md | 11 + CLAUDE.md | 2 + docs/architecture-invariants.md | 14 + docs/wiki/Settings-Reference.md | 1 + docs/wiki/The-Dashboard.md | 14 +- src/session-auto-name.ts | 384 +++++++++++++++++++----- src/session.ts | 60 +++- src/types/session.ts | 21 +- src/web/public/i18n.js | 1 + src/web/public/index.html | 7 + src/web/public/settings-ui.js | 3 + src/web/routes/session-routes.ts | 20 +- src/web/routes/ws-routes.ts | 3 +- src/web/schemas.ts | 7 + src/web/server.ts | 3 + src/web/session-listener-wiring.ts | 32 +- test/mocks/mock-session.ts | 3 + test/routes/session-name-routes.test.ts | 60 ++++ test/session-auto-name.test.ts | 274 +++++++++++++++++ test/session-listener-wiring.test.ts | 74 +++-- test/session-submit-anchor.test.ts | 29 -- 21 files changed, 874 insertions(+), 149 deletions(-) create mode 100644 .changeset/auto-name-sessions.md create mode 100644 test/routes/session-name-routes.test.ts create mode 100644 test/session-auto-name.test.ts diff --git a/.changeset/auto-name-sessions.md b/.changeset/auto-name-sessions.md new file mode 100644 index 00000000..8dbb6f95 --- /dev/null +++ b/.changeset/auto-name-sessions.md @@ -0,0 +1,11 @@ +--- +"aicodeman": patch +--- + +Auto-name sessions from the first prompt (#376, opt-in). With the new synced **Auto-name Sessions** setting on (App Settings → Appearance → Tabs, default off), a tab that still carries its generated name takes a title from the first real prompt you submit, keeping the case prefix: `w3-myapp` becomes `w3-myapp: fix the login redirect`. The strip shows the title with the prefix in the tooltip, and the next session in that case still counts up. It happens once per session, only for prompts you type or send through the input API (never a Ralph, respawn, cron or approval answer), never for shells, and a name you set yourself is never touched. Slash commands such as `/clear` do not become titles. The title is derived locally from the prompt's first sentence; no text leaves the machine. `nameSource` (`placeholder` / `auto` / `manual`) is a new additive field on session state. + +Landed with the fixes the review of #376 asked for: first prompt only (not every prompt), a user-input gate so Ralph, respawn, cron and approval writes cannot name a tab, the prefix form so the case identity and `w` counter survive, and a keystroke tracker that handles a bare Esc, bracketed pastes, wheel reports, Tab and history recall instead of mis-titling the tab. + +### Thanks + +- @shenlvkang-collab for #376, the auto-naming idea and the ownership plumbing (`nameSource`, the listener wiring, the restore path) it shipped with. diff --git a/CLAUDE.md b/CLAUDE.md index 7ef88de4..5401e6c8 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -233,6 +233,8 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph **Session lineage lines** (tab → tab it spawned, `sessionLineageLines`, per-device, desktop default ON): a create request may name the session that spawned it, as a `parentSessionId` body field on `POST /api/sessions` / `POST /api/quick-start` or the `X-Codeman-Parent-Session` header (the agent skill sets that once on its shared curl invocation, so every spawn recipe carries it). `resolveParentSessionId()` (route-helpers.ts) **resolves rather than trusts** it: exact id, else a UNIQUE ≥8-char prefix (ids reach agents truncated), it must be a live session the caller can see AND carry the same owner, and **anything unresolvable is DROPPED, never a 400** — a cosmetic field must not be able to fail a worker spawn. It rides `toState()` into `session_created`, so there is no new SSE event. ⚠️ Rendering is an ADDITIONAL LAYER on the existing SVG pass (`_appendLineageConnectionLines` called at the tail of `_updateConnectionLinesImmediate()`, exactly like ultracode), sharing one batched read→write reflow and the `tab:` rect cache; geometry is pure in `computeLineagePath()` (constants.js). ⚠️ **ONE shape, and the second one was the bug**: every pair (flat strip or wrapped) gets a U-bridge hanging below the strip, anchored on both tabs' BOTTOM edges. A wrapped strip used to get a parent-bottom → child-TOP bezier with a ~14px row gap to bend in, which drew a flat line hidden in the gap with siblings overprinting. ⚠️ The dip is a **mis-tuned-in-both-directions corridor** (44px cap = straight thread at strip-wide spans, #285; 104px cap + full row offset = ~106px over-bow into the terminal, 2026-08-15): it now hangs from the **STRIP's bottom edge** (fallback: lower tab bottom), capped at 64px, with NO per-row offsets stacked on top — the strip-bottom baseline is also what keeps a row-1 pair's arc from drawing through row 2's tab labels. ⚠️ **Colors are keyed on the SPAWNING tab, not per child**: every arc leaving one tab is the same color however many workers it spawns, so the strip reads as "these five came from w1, those two came from w2" — per-child coloring gave one tab's own children a different color each, which is the distinction the colors exist to make. A child that spawns in turn is a parent in its own right and gets its own color for the arcs below it, so a chain changes color at each generation while each generation's fan-out stays uniform. Assignment cycles `CodemanLineage.COLORS` in first-seen order per parent id (first entry empty = the skin-tuned `--session-blue`, so the first spawning tab keeps it; the rest vivid fixed hexes), memoized rather than derived from draw index (the SVG is wiped and rebuilt constantly, so an index-based color would flicker), and set inline as `--lineage-color` so styles.css keeps owning opacity/glow/dash. `test/session-lineage-lines.test.ts` drives the real `_appendLineageConnectionLines()` and asserts the painted property, since testing the color function alone would pass just as happily with the child id passed back in. ⚠️ **Desktop only**: the overlay is `z-index: 999` and the desktop header is 100 (arcs paint over it, which is what lets them touch tab bottoms), but under 1024px mobile.css makes the header `fixed; z-index: 1200` and would bury them. ⚠️ Paths carry `data-agent-id="lineage:"` because that is what `_applyLineEntrances()` queries — that one attribute is what gives them the entrance animation and its negative-`animation-delay` resume across `svg.innerHTML=''`. ⚠️ `.session-tabs` is `overflow-x: auto`, so a scrolled-out tab still HAS a rect (over the logo); edges with an endpoint outside the strip are skipped, and a passive `scroll` listener re-anchors the rest. +**Auto-named sessions** (`autoNameSessions`, SYNCED, default OFF; #376): a placeholder tab (`w3-myapp`) takes its first real prompt as a title, in the `: ` form (`w3-myapp: fix the login redirect`) that `parseSessionPrefix()` (app.js, #232) already renders as the title alone with the prefix in the tooltip and that `_nextCaseSessionStartNumber()` still counts, so the case identity and the `w<n>` counter survive. Ownership is the tri-state `SessionState.nameSource`: `placeholder` (Codeman's own `w<n>-<case>` or no name, inferred by `isGeneratedSessionName()` when a persisted state predates the field), `auto` (titled once), `manual` (the `name` setter, i.e. `PUT /api/sessions/:id/name`, which auto-naming never touches again). ⚠️ **First prompt means the FIRST**: `applyAutoName()` flips a placeholder to `auto` whether or not the string changed, so a later "1" cannot rename the tab; a prompt that yields no title (`/clear`, a `!` shell escape) leaves the session eligible for the next one. ⚠️ **Only user-originated input counts** (`SessionWriteOptions.fromUser`, set by the browser WS path and `POST /api/sessions/:id/input` ONLY, so a forgotten flag on a new path fails toward not naming): Ralph kick-starts, respawn `/clear`s, cron launches, approval answers and the trust-dialog keys write through the same `write()`/`writeViaMux()` and used to name every Ralph tab "Read @ralph_prompt.md…". A `startMode: 'shell'` CLI never feeds the tracker (a capability, not an id check; a shell tab was renamed after every `ls`), and the send-key route's Shift+Enter line feed bypasses the session entirely, so it calls `trackUserInput()` or the two lines join with no separator. ⚠️ **The tracker (`session-auto-name.ts`, pure) sits on the raw keystroke stream**, so every key has an explicit rule: a bare Esc is resolved at the END of the chunk it arrives in (it used to stay in escape mode and eat the next prompt's first character, or a whole CJK prompt); SGR mouse reports, Tab, cursor keys and Shift+Tab leave the draft alone (a wheel tick mid-word used to drop the first half); Up/Down and Ctrl+P/N/R TAINT the draft so Enter submits nothing rather than a fragment; bracketed-paste newlines are newlines IN the composer, never Enter. The title is the first sentence past a minimum length ("e.g. fix this now" is not "e.g."), capped at 72 code points, and the composed name honours `MAX_SESSION_NAME_LENGTH`. The listener (`session-listener-wiring.ts`) checks eligibility BEFORE reading the setting, so an already-named session costs no settings read per prompt. Opt-in because the prompt lands in the tab name, `mux-sessions.json`, every `session:updated` and `/api/search` (Read My Mind keeps prompts 0600 for the same reason). Tests: `test/session-auto-name.test.ts`, `test/session-listener-wiring.test.ts`, `test/routes/session-name-routes.test.ts`. → [architecture-invariants#auto-named-sessions-first-prompt--tab-title](docs/architecture-invariants.md#auto-named-sessions-first-prompt--tab-title) + **Maintainer bot (external)**: the Telegram bot that reviews open PRs and triages discussion threads in Codeman sessions used to live at `scripts/pr-bot/`. It moved OUT of this repository on 2026-09-14, to `~/codeman-cases/prbot/` (its own private git repo, systemd unit `codeman-pr-bot`, guide + agent rules in its own `README.md` and `CLAUDE.md`). It is a CLIENT of Codeman's HTTP API like any other, so nothing here depends on it and it is not part of the server, the CLI or the npm package. ⚠️ It spawns real sessions named `prbot-<n>` / `dscbot-<n>` on the local Codeman and holds clones under `~/.codeman/pr-bot/`, so those session names and that data dir are taken; it also fetches PR heads into `refs/pr-bot/*` of this checkout and must never check out, reset or clean it. The CHANGELOG entries for 1.25.0 and earlier still describe it, which is history rather than drift. **Unified session list**: `GET /api/sessions/unified` merges live sessions, persisted state, lifecycle-log history, and transcript files into one deduped list (pure core in `src/services/unified-session-service.ts`). ⚠️ **Transcript history is THREE stores, not one**, because each CLI keeps its conversations in its own: Claude's `~/.claude/projects`, omp's `~/.omp/agent/sessions` and codex's `~/.codex/sessions` (#386). Rows fold into their owning session via the `claudeSessionId → Codeman id` alias map, so resumed and `/clear`-respawned sessions do not appear twice; that field is named for Claude and carries whatever id the CLI names its conversation with, which for every non-Claude row diverges from the Codeman id by construction. ⚠️ **`resumeId` is set by a SCANNER row only, never by a live session**, and that is what makes it safe to resume on: a row carrying one is a conversation already on disk, so `resumeHistorySession()` sends `codexConfig.resumeSessionId` and a row without one is a genuinely fresh session. Every surface that re-projects these rows has to carry the field through, the phone overview included, or a tap on that surface silently starts a second conversation. No terminal buffers in the response, unlike `/api/sessions`. Backs the Cmd+K Session Manager, plus pinning and cross-device tab order (`PUT /api/session-order`; pure merge helpers in `src/session-order.ts`, pushing device wins and server-only ids are never dropped). → [architecture-invariants#unified-session-list-and-session-manager](docs/architecture-invariants.md#unified-session-list-and-session-manager) diff --git a/docs/architecture-invariants.md b/docs/architecture-invariants.md index eb9b772a..316c7fc3 100644 --- a/docs/architecture-invariants.md +++ b/docs/architecture-invariants.md @@ -114,6 +114,20 @@ Tests: `test/docker-hosts.test.ts`, `test/docker-exec-options.test.ts`, `test/do ⚠️ **`data-agent-id="lineage:<childId>"` is load-bearing**, not a label: `_applyLineEntrances()` queries paths by that attribute, so tagging them this way is the whole reason the arcs get the draw-in animation AND its negative-`animation-delay` resume across `svg.innerHTML = ''` with zero new animation code. ⚠️ `.session-tabs` is `overflow-x: auto`, so a tab scrolled out of the strip still HAS a rect — one lying over the logo or the header buttons; edges with an endpoint outside the strip are SKIPPED (clamping would point at a tab that is not there), and a passive `scroll` listener re-anchors the rest, since a scroll moves both endpoints without firing any render. The incremental tab render also redraws when `_lineageEdgeCount > 0`: a badge appearing widens a tab and shifts every tab after it. Setting: `sessionLineageLines`, per-device (in `displayKeys`, absent from the `.strict()` `SettingsUpdateSchema`), desktop default ON. Tests: `test/session-lineage-lines.test.ts` (geometry), `test/routes/session-routes-parent-lineage.test.ts` (resolution + reject paths). +### Auto-named sessions (first prompt → tab title) + +**Shipped opt-in, in the prefix form, after a review round that found five ways the first cut named a tab wrong** (#376, 1.30.0). The contributed version renamed on EVERY prompt (`applyAutoName` never left the eligible state, so "fix the login bug" then "1" left the tab named **1**), fed its tracker from every write path (shell tabs renamed after each command, every Ralph and respawn tab named "Read @ralph_prompt.md and follow the instructions."), stayed in escape mode after a bare Esc until a byte in `0x40-0x7e` arrived (Esc then "fix the login bug" submitted **ix the login bug**, Esc then a CJK prompt submitted nothing and the following prompt lost its first character, Esc then digits grew the escape buffer to 19001 characters), treated the newlines inside a bracketed paste as Enter, cleared the draft on ANY CSI (including the SGR wheel reports Codeman forwards to claude ≥ 2.1.187, so "fix the " + wheel + "login bug" gave **login bug**) and on Tab (the `@` completer), and replaced the whole name, which dropped the case from the tab and reset `_nextCaseSessionStartNumber()` so every new session in the case became `w1-<case>` again. Each of those is a named rule in `session-auto-name.ts` with a test. + +**Three owners, one setter.** `SessionState.nameSource` is `placeholder` | `auto` | `manual`. The constructor infers a missing value from the name (`isGeneratedSessionName()` = `w<n>-<case>` / `s<n>-<case>`, or no name at all, is a placeholder; anything else was a person's), the create routes pass none, the boot restore passes the persisted one. The `name` setter is the manual path and the ONLY thing that produces `manual` after construction; `applyAutoName()` is the only thing that produces `auto`, and it does so whether or not the string changed, which is what makes "first prompt" mean the first. A prompt whose title is null (`/clear`, `! npm test`, blank) never reaches it, so the session stays eligible: the first REAL prompt names the tab. + +**The origin gate defaults to "system".** `write()` / `writeViaMux()` take `SessionWriteOptions.fromUser`; only the browser WS path and `POST /api/sessions/:id/input` set it. Every other caller (Ralph, respawn, cron, approvals, the orchestrator's `/compact`, auto-ops, the trust-dialog keys) is system by omission, so a new user-input path that forgets the flag fails toward a tab that keeps its placeholder, never toward a tab named after a Ralph prompt. `_lastSubmitAt` is still stamped for every write; only the tracker feed is gated. The shell gate is `getCli(mode)?.capabilities.startMode !== 'shell'`, a capability rather than an id check (the no-id-branching guard), and the send-key route feeds `trackUserInput()` by hand because its `tmux send-keys -H` line feed never passes through the session. + +**The tracker is a best-effort transcript with explicit per-key rules**, not a byte filter. Mirrored: printable text, backspace, Ctrl+W, Ctrl+U/Ctrl+C (composer emptied), `\n` and Alt+Enter (a newline IN the composer, joined with a space), bracketed paste (newlines inside it likewise). Ignored: cursor keys, Home/End/Delete, Shift+Tab, Tab, SGR mouse and focus reports, Alt chords, OSC/DCS, the rest of C0. Tainting: Up/Down (CSI and SS3), Ctrl+P/N/R, Ctrl+_, because the composer then holds a history line the tracker never saw and Enter must submit nothing rather than a fragment. A bare Esc is resolved at the END of the chunk it arrives in, since xterm hands each key's whole sequence to one write and the programmatic senders send Esc alone; a CSI split across chunks still resumes. The draft keeps its HEAD past 8192 code points (the title is the first sentence, so keeping the tail would title a long paste by its last line) and an escape sequence is abandoned past 64 bytes. + +**Title and composition.** `deriveAutoSessionName()` strips CSI/control bytes, refuses slash commands by the `/^\/[a-z][a-z0-9_:-]*(\s|$)/i` shape (a path has a second slash where the whitespace should be, so `/home/me/notes.txt what is this` is a prompt) and `!` shell escapes, cuts at the first sentence terminator only past 8 code points ("e.g. fix this now" is not "e.g."), drops a trailing full stop, and caps at 72 code points on a word boundary. `composeAutoSessionName()` prepends the placeholder (`w3-myapp: fix the login redirect`) and fits the result into `MAX_SESSION_NAME_LENGTH` in UTF-16 units, the unit the rename route caps in. + +**Opt-in, and the listener orders its checks for cost.** The prompt lands in the tab name, `mux-sessions.json`, every `session:updated` broadcast, the TUI, both home screens and `/api/search` (which matches on `sessionName`), while Read My Mind deliberately keeps prompts 0600 and out of search because prompts can carry secrets; so `autoNameSessions` is synced and default OFF, like `agentSkillEnabled`, `approvalsInboxEnabled` and `readMyMindEnabled`. The listener checks `nameSource` and derives the title BEFORE reading `settings.json`, so an already-named session costs nothing per prompt. + ### Full-scrollback replay **Full-scrollback replay** (COD-164/#148, reworked for #205): `GET /api/sessions/:id/terminal?full=1` returns the ENTIRE tmux scrollback (capture-pane `-e -S -<lines>` bounded by the configured history limit, explicit `maxBuffer` from the terminal-history config, early byte-cap before normalization, CRLF-normalized for shell panes). On success the capture is returned ALONE (`source='mux-full-history'` — it supersedes the byte buffer; no duplication). The first load of each non-shell TUI session per page requests `full=1` (`_fullHistoryLoaded` Set in app.js — the old one-shot `_initialFullBufferLoad` flag was consumed by whichever tab auto-selected, leaving every other TUI tab one frame of history). Shell sessions instead load a bounded 1 MiB `?tail=` window on every selection and automatic drop recovery: a 100k-line shell capture can be tens of MiB, and automatically parsing it makes tab-switch latency scale with the entire session. Shell full history is explicit-button-only; reaching the top during an ordinary wheel/touch gesture must not reset xterm and replay the multi-megabyte capture on its main thread. Other modes may still re-pull `full=1` at the TOP, and pressing **Load full history** forces the request for any recoverably truncated session (`_maybeRefetchFullHistory`, 4s per-session gesture cooldown, in-flight + tab-switch guards, viewport position held across the replay); Shell full pulls are not retained in the tab cache, so the next switch stays bounded. Chunked replay enqueues 32 KiB pieces across safe yields, appends an xterm parse marker, then releases the live-output gate; output arriving after that release stays ordered behind the snapshot, while the marker callback supplies accurate parse timing without extending the pre-existing queued-event discard window. Live output is separately one-chunk-in-flight: xterm's callback releases each 32/64 KiB write before the next is submitted, keeping the remainder in the app queue where the 128 KiB cap can observe it instead of hiding an unbounded backlog in xterm's private WriteBuffer. While WebSocket owns terminal I/O, parallel SSE terminal/output-recovery events are discarded before JSON parsing; fallback recovery is single-flight per active session so backpressure cannot start overlapping reset+replay cycles. The route exposes capture/prepare totals in `Server-Timing`, while `[TERMINAL-PERF]` separates TTFB, body/JSON, reset+parse and total time for both selection and on-demand full pulls; parse completion is not a browser compositor/GPU paint measurement. The re-pull exists because xterm's buffer is only a WINDOW onto tmux's history and two things shrink it: tmux coalesces bursty output into pane REPAINTS that overwrite rows instead of emitting linefeeds (measured: a 60-line burst added 1 row of browser scrollback and destroyed 34), and a tab switch replays only the visible frame. tmux's own history is intact throughout — the browser just has to ask for it again. On-demand rather than automatic because at a 100k history limit the capture can be megabytes. ⚠️ **The capture ENDS with a cursor move back to the pane's own caret position** (`formatCursorRestore`, from the same `display-message` query the visible-frame path uses). The linear replay otherwise leaves the caret wherever the last character landed — the bottom-most row carrying text, which for an agent CLI is the status line — so the caret sat on the composer's border instead of its input line and every cursor-relative update the CLI sent afterwards was measured from the wrong row, until its next full redraw silently repaired it (that self-repair is why the report read as "it fixes itself as soon as Claude writes a line"). ⚠️ **The move is RELATIVE — up `rows - 1 - cursor_y`, then `\r`, then right `cursor_x` — never `CUP`.** `\x1b[<row>;<col>H` numbers rows from the top of the browser's screen, so it lands correctly only while the browser's row count equals `pane_height`, and nothing guarantees that: `resizeWindow` issues its tmux resize fire-and-forget and returns immediately, so a capture can be taken before a requested resize has applied, and `_onSessionNeedsRefresh` sends no resize at all. Counting up from the last replayed row anchors to the content both ends share. Restoring the cursor makes ROW ALIGNMENT load-bearing on this path: **no transform that can DELETE A LINE may run over a full-history capture**, because every deletion shifts the frame out from under the restored position. Four had accumulated — trailing blank rows stripped by `\n+$`, `stripInkRedrawBloat`, the `CLAUDE_BANNER_PATTERN` trim that cuts everything above the banner, and `LEADING_WHITESPACE_PATTERN` — each correct for a byte stream of successive frames and each wrong for a single rendered frame. ⚠️ **Those skips key on `isFullCapture`, meaning a capture actually came back — never on `?full=1` alone.** When `captureActivePaneBuffer` returns null (ENOBUFS, a timeout, a vanished pane, or a session with no mux at all) the reply falls back to `session.terminalBuffer`, which IS a byte stream and must still be stripped; gating on the query flag returned it whole, and a direct-PTY session takes that path on every first selection rather than only during an outage. ⚠️ A capture holding nothing visible (`hasVisibleContent`) returns `''`, because the caller reads an empty capture as "unavailable" and keeps its byte history — retaining trailing blank rows made an all-blank pane non-empty, which would have replaced real history with a blank screen from the server side, where `_replayWouldShrinkBuffer` cannot see it. ⚠️ **"One line per screen row" holds only where no row was hard-wrapped**: `-J` joins a wrapped row into its logical line (measured: a 100-character line in a 40-column pane captures as 10 lines against a 12-row pane), and the counts reconcile only once the browser xterm re-wraps at the same width — the same assumption `_estimateReplayRows` already documents. Tests: `test/tmux-capture-full-history.test.ts` covers the cursor move, the trim pairing and `hasVisibleContent`; `test/routes/session-routes.test.ts` covers a surviving blank first row, an unstripped byte-history fallback, and an empty capture leaving history intact. ⚠️ **The re-pull must never DOWNGRADE the buffer** (#205 round 2): the same reasoning that makes it a win for a shell pane makes it destructive for a repaint-mode CLI pane, where tmux keeps no history of its own (`history_size≈0` measured for a Claude pane) and the capture is roughly ONE frame while xterm may hold hundreds of rows of replayed frames — `_resetTerminalForReplay()` + rewrite then deletes history mid-scroll ("goes back a bit, repeats blocks, gets worse the further up I go"; measured A/B on a live pane: 341 rows → 42 with the guard off). `_replayWouldShrinkBuffer()` (terminal-ui.js) estimates the capture's rendered rows — escape sequences stripped, `capture-pane -J` re-wrapping accounted for — and the pull is skipped when that is more than one screen short of `buffer.active.length`. The one-screen tolerance matters: both sides are estimates (the buffer length counts trailing blank rows), so only a clear downgrade is refused. A refused session joins `_fullHistoryRepullUseless`, raising its cooldown from 4s to 60s so a hollow pane stops re-fetching megabytes on every scroll-up. Tests: `test/tmux-capture-full-history.test.ts`, `test/tmux-scrollback-eol.test.ts`, `test/terminal-scroll-routing.test.ts`, `test/terminal-flush-budget.test.ts`. diff --git a/docs/wiki/Settings-Reference.md b/docs/wiki/Settings-Reference.md index 6e8e31ed..a23a44c6 100644 --- a/docs/wiki/Settings-Reference.md +++ b/docs/wiki/Settings-Reference.md @@ -76,6 +76,7 @@ every session or only the active tab. | Tall Tabs | Taller tab strip. | | Pop-out Button on Tabs | Adds the detach control to tabs, with a per-tab override. | | Spawn Lineage Lines | Arcs from a parent tab to sessions it spawned. Desktop only, on by default. | +| Auto-name Sessions | Titles a new tab after its first prompt, keeping the case prefix (`w3-myapp: fix the login redirect`). Synced, off by default. See [The Dashboard](The-Dashboard#automatic-session-names). | | Overview Home Screen | The phone home screen. On by default. | ### Models diff --git a/docs/wiki/The-Dashboard.md b/docs/wiki/The-Dashboard.md index f6c2d171..5672b70a 100644 --- a/docs/wiki/The-Dashboard.md +++ b/docs/wiki/The-Dashboard.md @@ -68,11 +68,15 @@ Tabs can also be dragged to reorder. ### Automatic session names -New sessions start with a short project/sequence name so they can be created immediately. -After the first task prompt is submitted, Codeman replaces that placeholder with a short -title derived locally from the prompt's first sentence. Slash commands such as `/clear` do -not become titles. A name you set with the inline rename action is treated as manual and is -never overwritten by automatic naming. +Off by default. Turn on **Auto-name Sessions** (App Settings → Appearance → Tabs; synced +across devices) and a tab that still carries its generated name, such as `w3-myapp`, takes a +title from the first real prompt you submit, keeping the prefix: `w3-myapp: fix the login +redirect`. The strip shows the title and keeps the prefix in the tooltip, and the next +session in that case still counts up to `w4-myapp`. It happens once per session, only for +prompts you type or send through the input API (never a Ralph, respawn, cron or approval +answer), and never for shells. Slash commands such as `/clear` do not become titles; the +next prompt gets its turn. A name you set yourself, before or after, is never touched. The +title is derived locally from the prompt's first sentence; no text leaves the machine. On phones the strip scrolls horizontally instead of wrapping, and the active tab is always scrolled into view. It is not reordered to the front, so the `Alt+N` numbering stays stable. diff --git a/src/session-auto-name.ts b/src/session-auto-name.ts index 82c57890..fa3f1448 100644 --- a/src/session-auto-name.ts +++ b/src/session-auto-name.ts @@ -1,101 +1,347 @@ /** - * Helpers for assigning a useful default name after the first submitted prompt. + * @fileoverview Automatic session names from the first prompt. * - * This deliberately does not call an LLM: the prompt is already available at - * the input boundary, so a bounded local title is private, deterministic, and - * works for every CLI backend. + * A new tab is born as `w3-myapp`, which says where it runs and nothing about + * what it is doing. Once the user submits a real prompt the tab can carry a + * title derived from it (`w3-myapp: fix the login redirect`), and this module + * holds the three pure pieces of that: a tracker that reconstructs the composer + * text from the keystrokes Codeman forwards, the title heuristic, and the + * prefix-preserving composition. + * + * Deliberately no LLM: the prompt already passes through the input boundary, + * so a local title is private, deterministic and identical for every CLI. + * + * ⚠️ The tracker sits on the raw keystroke stream, which carries far more than + * the prompt: cursor keys, mouse reports Codeman forwards to the CLI, bracketed + * pastes, Alt chords, the bare Esc that interrupts a turn. Every one of those + * once named a tab something wrong (a lone Esc ate the next prompt's first + * character; a wheel tick mid-word dropped the first half of the prompt), so + * the rules below are explicit per key. The model is a best-effort transcript: + * keys whose effect on the composer is knowable are mirrored, keys that leave + * the text alone are ignored, and keys that replace it with something the + * tracker cannot see (history recall) TAINT the draft so that Enter submits + * nothing rather than a fragment. A prompt that yields no title leaves the + * session eligible for the next one. + * + * Only user-originated input is fed here; the Session decides that. Ralph + * kick-starts, respawn `/clear`s, cron launches and approval answers all go + * through the same write paths and must never become a tab title. + * + * @module session-auto-name */ -const MAX_PROMPT_BUFFER_LENGTH = 8_192; +import { MAX_SESSION_NAME_LENGTH } from './config/terminal-limits.js'; + +/** + * Longest composer draft kept, in code points. The title is cut from the HEAD + * of the prompt, so once the cap is reached further text is counted rather + * than kept (backspaces consume that count first). Keeping the tail instead + * would turn a long paste into a title made of its last line. + */ +const MAX_PROMPT_BUFFER_CODE_POINTS = 8_192; + +/** Longest escape sequence collected before the tracker gives up on it. */ +const MAX_ESCAPE_SEQUENCE_LENGTH = 64; + +/** Longest title, in code points, before it is cut with an ellipsis. */ const MAX_AUTO_NAME_CODE_POINTS = 72; /** - * Tracks terminal input until Enter is received. Terminal input arrives in - * arbitrary chunks, so this keeps only a small composer buffer and ignores - * navigation/control escape sequences. + * A sentence boundary is only honoured this far into the prompt, or "e.g. fix + * this now" becomes "e.g." and "Ok. Fix the bug" becomes "Ok". Short enough + * that a CJK sentence (a dozen code points is a full request) still cuts. + */ +const MIN_SENTENCE_CODE_POINTS = 8; + +/** A CSI sequence ends at its first byte in this range. */ +const CSI_FINAL_BYTE = /[\x40-\x7e]/; +/** CSI parameter and intermediate bytes; anything else mid-sequence is malformed. */ +const CSI_BODY_BYTE = /[\x20-\x3f]/; + +/** + * `/clear`, `/model opus`, `/ralph-loop:ralph-loop`: a slash followed by a + * command word and then whitespace or the end. A path (`/home/me/notes.txt + * what is this`) has a second slash where the whitespace should be and so is a + * prompt. + */ +const SLASH_COMMAND_PATTERN = /^\/[a-z][a-z0-9_:-]*(?:\s|$)/i; + +// eslint-disable-next-line no-control-regex +const CSI_SEQUENCE_PATTERN = /\x1b\[[\x30-\x3f]*[\x20-\x2f]*[\x40-\x7e]/g; +// eslint-disable-next-line no-control-regex +const CONTROL_CHAR_PATTERN = /[\x00-\x1f\x7f]/g; +const SENTENCE_TERMINATORS = new Set(['.', '!', '?', '。', '!', '?']); + +/** + * Reconstructs the composer draft from forwarded keystrokes and reports each + * submitted prompt. Input arrives in arbitrary chunks (one keystroke, a paste, + * an agent's whole prompt plus Enter), so all state lives across calls. */ export class SubmittedPromptTracker { private buffer = ''; - private escapeSequence = ''; + private bufferCodePoints = 0; + /** Code points typed past the cap; backspaces eat these before real text. */ + private overflow = 0; + /** Escape sequence in progress; a lone ESC means "just saw ESC". */ + private sequence = ''; + private inPaste = false; + /** The composer holds text the tracker never saw (history recall); Enter submits nothing. */ + private tainted = false; feed(data: string): string[] { const submitted: string[] = []; - - for (const character of data) { - if (this.escapeSequence) { - this.escapeSequence += character; - // CSI sequences end with a byte in the final-byte range. - const isCsiIntroducer = this.escapeSequence === '\x1b[' || this.escapeSequence === '\x1bO'; - if (/[\x40-\x7e]/.test(character) && !isCsiIntroducer) { - const isBracketedPasteMarker = this.escapeSequence === '\x1b[200~' || this.escapeSequence === '\x1b[201~'; - if (!isBracketedPasteMarker) this.buffer = ''; - this.escapeSequence = ''; - } + for (const ch of data) { + if (this.sequence) { + this.continueSequence(ch); continue; } - - if (character === '\x1b') { - this.escapeSequence = character; + if (ch === '\x1b') { + this.sequence = ch; continue; } + this.handleKey(ch, submitted); + } + // A chunk that ENDS in a lone ESC is the Esc key, not the start of a + // sequence: xterm hands each key's whole sequence to one write, and the + // programmatic senders (an approval deny sends exactly `\x1b`) send it + // alone. Leaving it pending would make the next prompt's first character + // look like an Alt chord and swallow it. + if (this.sequence === '\x1b') this.sequence = ''; + return submitted; + } - if (character === '\r' || character === '\n') { - const prompt = this.buffer.trim(); - if (prompt) submitted.push(prompt); - this.buffer = ''; - continue; - } - - if (character === '\x08' || character === '\x7f') { - this.buffer = Array.from(this.buffer).slice(0, -1).join(''); - continue; - } - - const codePoint = character.codePointAt(0) ?? 0; - if (codePoint < 0x20 || codePoint === 0x7f) { - // Ctrl-C/Ctrl-U and cursor controls make the append-only buffer - // unreliable. The next printable text starts a fresh candidate. - this.buffer = ''; - continue; - } - - this.buffer += character; - if (this.buffer.length > MAX_PROMPT_BUFFER_LENGTH) { - this.buffer = this.buffer.slice(-MAX_PROMPT_BUFFER_LENGTH); + private continueSequence(ch: string): void { + if (this.sequence === '\x1b') { + if (ch === '[' || ch === 'O' || ch === ']' || ch === 'P') { + this.sequence += ch; + return; } + this.sequence = ch === '\x1b' ? ch : ''; + // Alt+Enter inserts a newline in the composer; every other Alt chord + // (word movement, Alt+B/F) leaves the text alone. + if (ch === '\r' || ch === '\n') this.appendSeparator(); + return; } - return submitted; + this.sequence += ch; + if (this.sequence.length > MAX_ESCAPE_SEQUENCE_LENGTH) { + // Not a sequence any terminal sends; what follows is unknowable, so the + // draft is tainted rather than titled after the tail of the garbage. + this.sequence = ''; + this.tainted = true; + return; + } + + const kind = this.sequence[1]; + if (kind === '[') { + if (CSI_FINAL_BYTE.test(ch)) { + const sequence = this.sequence; + this.sequence = ''; + this.handleCsi(sequence); + } else if (!CSI_BODY_BYTE.test(ch)) { + // Malformed (an ESC [ followed by text): drop the sequence and let the + // character count as typed rather than swallowing up to 64 of them. + this.sequence = ''; + this.handleKeyOrEscape(ch); + } + return; + } + if (kind === 'O') { + // SS3 carries exactly one byte (application-mode cursor keys). + this.sequence = ''; + if (ch === 'A' || ch === 'B') this.tainted = true; + return; + } + // OSC / DCS run to BEL or ST (ESC \). + if (ch === '\x07' || this.sequence.endsWith('\x1b\\')) this.sequence = ''; + } + + private handleKeyOrEscape(ch: string): void { + if (ch === '\x1b') { + this.sequence = ch; + return; + } + // Only reached mid-chunk from a malformed sequence, where no submission can + // be reported; a stray Enter there resets the draft like any other Enter. + this.handleKey(ch, []); + } + + private handleCsi(sequence: string): void { + if (sequence === '\x1b[200~') { + this.inPaste = true; + return; + } + if (sequence === '\x1b[201~') { + this.inPaste = false; + return; + } + const final = sequence[sequence.length - 1]; + // Up/Down (with or without modifiers) recall history: the composer now + // holds a line this tracker never saw. Everything else leaves the text as + // it is: Left/Right/Home/End, Delete (`3~`), Shift+Tab (`Z`), SGR mouse + // reports (`<…M`/`m`, forwarded on every wheel tick), focus reports. + if (final === 'A' || final === 'B') this.tainted = true; + } + + private handleKey(ch: string, submitted: string[]): void { + const codePoint = ch.codePointAt(0) ?? 0; + if (this.inPaste) { + // Pasted newlines are newlines IN the composer, never Enter; they and + // the other controls (tabs) become a single separator. + if (codePoint < 0x20 || codePoint === 0x7f) this.appendSeparator(); + else this.append(ch); + return; + } + switch (ch) { + case '\r': { + const prompt = this.tainted ? '' : this.buffer.trim(); + if (prompt) submitted.push(prompt); + this.reset(); + return; + } + case '\n': + // Ctrl+J, and the line feed the send-key route injects for Shift+Enter: + // a newline inside the composer, so the lines join with a separator. + this.appendSeparator(); + return; + case '\x7f': + case '\x08': + this.backspace(); + return; + case '\x17': // Ctrl+W: word rubout + this.killWord(); + return; + case '\x15': // Ctrl+U: line discard + case '\x03': // Ctrl+C: clears the composer (or, empty, arms an exit) + this.reset(); + return; + case '\x10': // Ctrl+P + case '\x0e': // Ctrl+N + case '\x12': // Ctrl+R: history search + case '\x1f': // Ctrl+_: undo + this.tainted = true; + return; + default: + // Tab (the @-mention completer, which only ever extends the token), + // cursor chords (Ctrl+A/E/B/F) and the rest of C0 leave the text alone. + if (codePoint < 0x20 || codePoint === 0x7f) return; + this.append(ch); + } + } + + /** One space between lines, never a run of them, and none at the start. */ + private appendSeparator(): void { + if (this.overflow > 0) return; + if (!this.buffer || /\s$/.test(this.buffer)) return; + this.append(' '); + } + + private append(ch: string): void { + if (this.bufferCodePoints >= MAX_PROMPT_BUFFER_CODE_POINTS) { + this.overflow += 1; + return; + } + this.buffer += ch; + this.bufferCodePoints += 1; + } + + private backspace(): void { + if (this.overflow > 0) { + this.overflow -= 1; + return; + } + if (!this.buffer) return; + const last = this.buffer.charCodeAt(this.buffer.length - 1); + const units = last >= 0xdc00 && last <= 0xdfff && this.buffer.length >= 2 ? 2 : 1; + this.buffer = this.buffer.slice(0, -units); + this.bufferCodePoints -= 1; + } + + private killWord(): void { + this.overflow = 0; + this.buffer = this.buffer.replace(/\S+\s*$/u, ''); + this.bufferCodePoints = Array.from(this.buffer).length; + } + + private reset(): void { + this.buffer = ''; + this.bufferCodePoints = 0; + this.overflow = 0; + this.tainted = false; } } /** - * Converts a submitted prompt into a compact session title. - * Returns null for empty text and slash commands, which are usually controls - * such as /clear or /resume rather than the task the user wants to remember. + * Turns a submitted prompt into a title, or null when the prompt is not a task: + * empty, a slash command (`/clear`, `/model`), or a `!` shell escape. */ export function deriveAutoSessionName(prompt: string): string | null { - // Terminal input can legitimately contain ANSI/control bytes; they are - // removed before the title is persisted or broadcast. - const normalized = prompt - // eslint-disable-next-line no-control-regex - .replace(/\x1b\[[0-?]*[ -/]*[@-~]/g, '') - // eslint-disable-next-line no-control-regex - .replace(/[\u0000-\u001f\u007f]/g, ' ') - .replace(/\s+/g, ' ') - .trim(); - if (!normalized || normalized.startsWith('/')) return null; - - const firstSentence = normalized.match(/^.*?(?:[.!?。!?](?:\s|$)|$)/)?.[0]?.trim() || normalized; - const codePoints = Array.from(firstSentence); - if (codePoints.length <= MAX_AUTO_NAME_CODE_POINTS) return firstSentence; - return `${codePoints - .slice(0, MAX_AUTO_NAME_CODE_POINTS - 1) - .join('') - .trimEnd()}…`; + const text = prompt.replace(CSI_SEQUENCE_PATTERN, '').replace(CONTROL_CHAR_PATTERN, ' ').replace(/\s+/g, ' ').trim(); + if (!text || text.startsWith('!') || SLASH_COMMAND_PATTERN.test(text)) return null; + return truncateCodePoints(firstSentence(text), MAX_AUTO_NAME_CODE_POINTS); } -/** Existing Codeman-generated tab names are safe to upgrade on first prompt. */ +/** + * The first sentence, provided it is long enough to be one; a trailing full + * stop is dropped because a tab title is not a sentence. + */ +function firstSentence(text: string): string { + const codePoints = Array.from(text); + for (let i = MIN_SENTENCE_CODE_POINTS - 1; i < codePoints.length; i++) { + if (!SENTENCE_TERMINATORS.has(codePoints[i])) continue; + const next = codePoints[i + 1]; + if (next !== undefined && !/\s/.test(next)) continue; + return codePoints + .slice(0, i + 1) + .join('') + .replace(/[.。]+$/, ''); + } + return text.replace(/[.。]+$/, ''); +} + +/** Cuts to `max` code points with an ellipsis, on a word boundary when one is near the end. */ +function truncateCodePoints(text: string, max: number): string { + const codePoints = Array.from(text); + if (codePoints.length <= max) return text; + let cut = codePoints.slice(0, max - 1).join(''); + const lastSpace = cut.lastIndexOf(' '); + if (lastSpace >= Math.floor(cut.length / 2)) cut = cut.slice(0, lastSpace); + return `${cut.trimEnd()}…`; +} + +/** + * The name a placeholder becomes: `<prefix>: <title>`, so the tab keeps its + * case identity and its `w<n>` counter (the tab strip already renders that + * form as the title alone, prefix in the tooltip, and the next-session counter + * still matches it). A session with no name at all just takes the title. The + * result honours `maxLength` in UTF-16 units, the unit the rename route caps. + */ +export function composeAutoSessionName( + currentName: string, + title: string, + maxLength = MAX_SESSION_NAME_LENGTH +): string { + const prefix = currentName.trim(); + if (!prefix) return fitTitle(title, maxLength); + const room = maxLength - prefix.length - 2; + if (room <= 0) return prefix; + return `${prefix}: ${fitTitle(title, room)}`; +} + +/** Fits a title into `maxUnits` UTF-16 units, ellipsis included. */ +function fitTitle(title: string, maxUnits: number): string { + if (title.length <= maxUnits) return title; + let units = 0; + let keep = 0; + for (const codePoint of Array.from(title)) { + if (units + codePoint.length > maxUnits - 1) break; + units += codePoint.length; + keep += 1; + } + return truncateCodePoints(title, keep + 1); +} + +/** Codeman's own `w<n>-<case>` / `s<n>-<case>` placeholders, the only names auto-naming replaces. */ export function isGeneratedSessionName(name: string): boolean { return /^[ws]\d+-[a-zA-Z0-9_-]+$/.test(name); } diff --git a/src/session.ts b/src/session.ts index c5d8b034..21d5ac89 100644 --- a/src/session.ts +++ b/src/session.ts @@ -59,6 +59,7 @@ import { type SessionRemote, type SessionDocker, type SessionNameSource, + type SessionWriteOptions, } from './types.js'; import { resolveAndClaimOmpSessionId } from './utils/omp-session-resolver.js'; import { probeDockerCliVersion } from './docker-hosts.js'; @@ -426,7 +427,14 @@ export class Session extends EventEmitter { private _name: string; private _nameSource: SessionNameSource; + /** + * Reconstructs the composer draft from USER keystrokes so the first real + * prompt can name the tab. Fed only when a write says `fromUser`, and never + * for a CLI whose Enter runs a command rather than submitting a prompt + * (`startMode: 'shell'`), so a shell tab is not renamed after every `ls`. + */ private readonly _submittedPromptTracker = new SubmittedPromptTracker(); + private readonly _acceptsPrompts: boolean; private ptyProcess: pty.IPty | null = null; private _pid: number | null = null; private _status: SessionStatus = 'idle'; @@ -658,7 +666,11 @@ export class Session extends EventEmitter { workingDir: string; mode?: SessionMode; name?: string; - /** Whether the current name is still eligible for automatic replacement. */ + /** + * Who owns the name (see `SessionNameSource`). Omitted, it is inferred + * from the name: Codeman's own `w<n>-<case>` placeholders (or no name) + * stay eligible for auto-naming, anything else counts as the user's. + */ nameSource?: SessionNameSource; /** Terminal multiplexer instance (tmux) */ mux?: TerminalMultiplexer; @@ -730,7 +742,8 @@ export class Session extends EventEmitter { this.mode = config.mode || 'claude'; this._name = config.name || ''; this._nameSource = - config.nameSource ?? (!this._name || isGeneratedSessionName(this._name) ? 'auto' : 'manual'); + config.nameSource ?? (!this._name || isGeneratedSessionName(this._name) ? 'placeholder' : 'manual'); + this._acceptsPrompts = getCli(this.mode)?.capabilities.startMode !== 'shell'; this._resumeSessionId = config.resumeSessionId; // NOW, not `createdAt`: recovery passes the ORIGINAL creation time of a // days-old tmux session, and seeding last-activity from it would report a @@ -1378,15 +1391,25 @@ export class Session extends EventEmitter { return this._name; } + /** An explicit rename: the name is the user's from here on and auto-naming never touches it. */ set name(value: string) { this._name = value; this._nameSource = 'manual'; } - /** Replace an automatically generated name without taking ownership from auto naming. */ + /** + * Names the tab after its first prompt. Only a placeholder is eligible, and + * the session stops being one whether or not the string changed: "first + * prompt" means the first, not "every prompt until a rename". Returns + * whether the name changed, so the caller knows whether to persist and + * broadcast. + */ applyAutoName(value: string): boolean { + if (this._nameSource !== 'placeholder') return false; const name = value.trim(); - if (!name || this._nameSource !== 'auto' || this._name === name) return false; + if (!name) return false; + this._nameSource = 'auto'; + if (this._name === name) return false; this._name = name; return true; } @@ -3535,8 +3558,8 @@ export class Session extends EventEmitter { * discards the data, but it used to do so with no signal at all — which is how * input could disappear while the caller believed it had been delivered. */ - write(data: string): boolean { - const submittedPrompt = this._trackSubmit(data); + write(data: string, options: SessionWriteOptions = {}): boolean { + const submittedPrompt = this._trackSubmit(data, options); if (!this.ptyProcess) return false; this.ptyProcess.write(data); this._emitSubmittedPrompt(submittedPrompt); @@ -3556,14 +3579,31 @@ export class Session extends EventEmitter { return this._lastSubmitAt; } - private _trackSubmit(data: string): string[] { - const submitted = this._submittedPromptTracker.feed(data); + /** + * Stamps the pane's last Enter for EVERY write, and feeds the auto-name + * tracker only for user-originated input on a prompt-taking CLI. Ralph + * kick-starts, respawn `/clear`s, cron launches, approval answers and the + * trust-dialog keys all arrive without `fromUser` and so can never name a tab. + */ + private _trackSubmit(data: string, options: SessionWriteOptions): string[] { + const submitted = options.fromUser && this._acceptsPrompts ? this._submittedPromptTracker.feed(data) : []; if (data.includes('\r') || data.includes('\n')) { this._lastSubmitAt = Date.now(); } return submitted; } + /** + * Feeds user input that reaches the pane AROUND the write paths: the + * send-key route injects Shift+Enter's line feed through `tmux send-keys -H` + * directly, and without this the two lines of a prompt joined with no + * separator. Reports submissions like a write would (a line feed never is one). + */ + trackUserInput(data: string): void { + if (!this._acceptsPrompts) return; + this._emitSubmittedPrompt(this._submittedPromptTracker.feed(data)); + } + private _emitSubmittedPrompt(prompts: string[]): void { for (const prompt of prompts) { this.emit('promptSubmitted', prompt); @@ -3664,8 +3704,8 @@ export class Session extends EventEmitter { * session.writeViaMux('/init\r'); // Send /init command * ``` */ - async writeViaMux(data: string): Promise<boolean> { - const submittedPrompt = this._trackSubmit(data); + async writeViaMux(data: string, options: SessionWriteOptions = {}): Promise<boolean> { + const submittedPrompt = this._trackSubmit(data, options); if (this._mux && this._muxSession) { const sent = await this._mux.sendInput(this.id, data); if (sent) this._emitSubmittedPrompt(submittedPrompt); diff --git a/src/types/session.ts b/src/types/session.ts index e0eca362..c993666f 100644 --- a/src/types/session.ts +++ b/src/types/session.ts @@ -58,8 +58,23 @@ export type SessionMode = | 'deepseek' | 'omp'; -/** Whether a session name may still be replaced by the first submitted prompt. */ -export type SessionNameSource = 'auto' | 'manual'; +/** + * Who owns a session's name. `placeholder`: Codeman's own `w<n>-<case>` (or no + * name at all), still eligible for auto-naming. `auto`: titled after its first + * prompt (`w<n>-<case>: <title>`), which happens once. `manual`: set by a + * person; auto-naming never touches it. + */ +export type SessionNameSource = 'placeholder' | 'auto' | 'manual'; + +/** Options for `Session.write()` / `Session.writeViaMux()`. */ +export interface SessionWriteOptions { + /** + * The bytes were typed by a person, or sent by an agent on their behalf + * (browser keystrokes, `POST /api/sessions/:id/input`). Only such input can + * name a tab; Ralph, respawn, cron and approval writes leave this unset. + */ + fromUser?: boolean; +} export type RemoteCommandMode = Extract< SessionMode, @@ -617,7 +632,7 @@ export interface SessionState { lastActivityAt: number; /** Session display name */ name?: string; - /** Name ownership; auto names are replaced after the first real prompt. */ + /** Who owns the name (see `SessionNameSource`); absent on states persisted before auto-naming existed. */ nameSource?: SessionNameSource; /** Session mode */ mode?: SessionMode; diff --git a/src/web/public/i18n.js b/src/web/public/i18n.js index 23605d2f..bee8fcab 100644 --- a/src/web/public/i18n.js +++ b/src/web/public/i18n.js @@ -252,6 +252,7 @@ 'Ultracode Agents': 'Ultracode 智能体', 'Ultracode Floating Windows': 'Ultracode 浮动窗口', 'Approvals Inbox': '审批收件箱', + 'Auto-name Sessions': '自动命名会话', Approvals: '审批', 'Prompts waiting on you, across all sessions': '所有会话中等待您处理的提示', 'No pending approvals': '没有待处理的审批', diff --git a/src/web/public/index.html b/src/web/public/index.html index 39e956f1..2ba80243 100644 --- a/src/web/public/index.html +++ b/src/web/public/index.html @@ -2047,6 +2047,13 @@ </div> <label class="switch switch-sm"><input type="checkbox" id="appSettingsLineageLines" checked><span class="slider"></span></label> </div> + <div class="set-row" id="appSettingsAutoNameSessionsItem" data-search="auto name session title first prompt tab rename"> + <div class="set-row-text"> + <span class="set-row-label">Auto-name Sessions <span class="set-tag">synced</span></span> + <span class="set-row-desc">Title a new tab after its first prompt, keeping the case prefix. Renamed tabs are never touched.</span> + </div> + <label class="switch switch-sm"><input type="checkbox" id="appSettingsAutoNameSessions"><span class="slider"></span></label> + </div> <div class="set-row" id="appSettingsMobileOverviewItem" data-search="overview home screen phone logo"> <div class="set-row-text"> <span class="set-row-label">Overview Home Screen <span class="set-tag">phone</span></span> diff --git a/src/web/public/settings-ui.js b/src/web/public/settings-ui.js index 92631b0f..79bee3c0 100644 --- a/src/web/public/settings-ui.js +++ b/src/web/public/settings-ui.js @@ -408,6 +408,8 @@ Object.assign(CodemanApp.prototype, { // header), so the row is hidden elsewhere rather than offering a toggle that // changes nothing. Default ON — only an explicit false turns it off. document.getElementById('appSettingsLineageLines').checked = settings.sessionLineageLines ?? defaults.sessionLineageLines ?? true; + // Auto-name sessions: synced, default OFF (opt-in; only an explicit true enables). + document.getElementById('appSettingsAutoNameSessions').checked = settings.autoNameSessions === true; const lineageItem = document.getElementById('appSettingsLineageLinesItem'); if (lineageItem) lineageItem.style.display = MobileDetection.getDeviceType() === 'desktop' ? '' : 'none'; document.getElementById('appSettingsMobileOverview').checked = settings.mobileOverviewEnabled ?? defaults.mobileOverviewEnabled ?? false; @@ -2111,6 +2113,7 @@ Object.assign(CodemanApp.prototype, { showRedrawButton: document.getElementById('appSettingsShowRedrawButton').checked, mobileOverviewEnabled: document.getElementById('appSettingsMobileOverview').checked, sessionLineageLines: document.getElementById('appSettingsLineageLines').checked, + autoNameSessions: document.getElementById('appSettingsAutoNameSessions').checked, showSessionButton: document.getElementById('appSettingsShowSessionButton').checked, showAwayDigestButton: document.getElementById('appSettingsShowAwayDigestButton').checked, showCronButton: document.getElementById('appSettingsShowCronButton').checked, diff --git a/src/web/routes/session-routes.ts b/src/web/routes/session-routes.ts index 950c04ff..03f92e5d 100644 --- a/src/web/routes/session-routes.ts +++ b/src/web/routes/session-routes.ts @@ -1081,7 +1081,6 @@ export function registerSessionRoutes( workingDir, mode, name: body.name || '', - nameSource: body.name ? undefined : 'auto', mux: ctx.mux, useMux: true, niceConfig: globalNice, @@ -1596,6 +1595,9 @@ export function registerSessionRoutes( // Write input to PTY. Direct write is synchronous; writeViaMux // (tmux send-keys) is fire-and-forget to avoid blocking the HTTP response. + // Every write here is `fromUser`: this route carries a person's prompt, or an + // agent's on their behalf, so it may name the tab (Ralph, respawn, cron and + // approvals write through the session directly and never say so). // // Because the response has already been sent by then, a failure there is the // one case the caller can never learn about — so the dedup bookkeeping is @@ -1618,32 +1620,32 @@ export function registerSessionRoutes( } else if (useMux && waitPromise) { // The response is already staying open for the wait, so the tmux write can be // awaited here. This is the ONE path where a writeViaMux failure is observable. - const ok = await session.writeViaMux(inputStr).catch(() => false); + const ok = await session.writeViaMux(inputStr, { fromUser: true }).catch(() => false); if (ok) { delivered = true; } else { console.warn(`[Server] writeViaMux failed for session ${id}, falling back to direct write`); - delivered = session.write(inputStr); + delivered = session.write(inputStr, { fromUser: true }); if (!delivered) undoOnFailure(); } } else if (useMux) { // Fire-and-forget: don't block the HTTP response on a tmux child process. // Fallback to a direct write on failure. Unchanged from before send-and-wait. session - .writeViaMux(inputStr) + .writeViaMux(inputStr, { fromUser: true }) .then((ok) => { if (ok) return; console.warn(`[Server] writeViaMux failed for session ${id}, falling back to direct write`); - if (!session.write(inputStr)) undoOnFailure(); + if (!session.write(inputStr, { fromUser: true })) undoOnFailure(); }) .catch(() => { - if (!session.write(inputStr)) undoOnFailure(); + if (!session.write(inputStr, { fromUser: true })) undoOnFailure(); }); } else { // Same rollback. NOT an error response, deliberately: a session can // legitimately have no PTY yet (created but not started), and callers have // always been able to write to one without a 4xx. - delivered = session.write(inputStr); + delivered = session.write(inputStr, { fromUser: true }); if (!delivered && tagged) { session.forgetInputSeq(clientId as string, seq as number); } @@ -1892,6 +1894,9 @@ export function registerSessionRoutes( console.error('[Server] send-key failed:', err); return createErrorResponse(ApiErrorCode.INTERNAL_ERROR, 'tmux send-keys failed'); } + // The bytes bypassed the session's write path, so tell the auto-name + // tracker about them or the two lines of a prompt join with no separator. + session.trackUserInput(hex.map((byte) => String.fromCharCode(parseInt(byte, 16))).join('')); return {}; }); @@ -3485,7 +3490,6 @@ export function registerSessionRoutes( const session = new Session({ workingDir: resolvedCasePath, name: sessionName ? sessionName.slice(0, MAX_SESSION_NAME_LENGTH) : '', - nameSource: sessionName ? undefined : 'auto', mux: ctx.mux, useMux: true, mode: mode, diff --git a/src/web/routes/ws-routes.ts b/src/web/routes/ws-routes.ts index ed60f5d3..ec7fe9e1 100644 --- a/src/web/routes/ws-routes.ts +++ b/src/web/routes/ws-routes.ts @@ -185,7 +185,8 @@ export function registerWsRoutes(app: FastifyInstance, ctx: SessionPort, getHost // Typed input from a claim-holding desktop keeps the claim "hot" // and re-asserts the desktop layout after a mobile override. if (holdsDesktopClaim) session.noteDesktopActivity(); - delivered = session.write(msg.d); + // Browser keystrokes are the user's own, so they may name the tab. + delivered = session.write(msg.d, { fromUser: true }); // A session whose PTY is gone swallows the write. ACKing anyway told // the client to drop the frame from its durable queue and left the seq // burnt, so the retry that reliable delivery exists for was rejected as diff --git a/src/web/schemas.ts b/src/web/schemas.ts index e7e8c28c..b5dabdd0 100644 --- a/src/web/schemas.ts +++ b/src/web/schemas.ts @@ -1230,6 +1230,13 @@ export const SettingsUpdateSchema = z * already pending immediately. */ approvalsInboxEnabled: z.boolean().optional(), + /** + * Auto-name sessions: a placeholder tab (`w3-case`) takes its first real + * prompt as a title (`w3-case: fix the login redirect`). Synced, default + * OFF: the prompt lands in mux-sessions.json, every session:updated + * broadcast and /api/search, which is the user's choice to make. + */ + autoNameSessions: z.boolean().optional(), /** * Read My Mind (docs/readmymind-plan.md): capture the user's submitted * prompts into per-case intent profiles. SYNCED, default OFF (opt-in: diff --git a/src/web/server.ts b/src/web/server.ts index 477ae5d6..cf5927c0 100644 --- a/src/web/server.ts +++ b/src/web/server.ts @@ -1729,6 +1729,9 @@ export class WebServer extends EventEmitter { registerAttachment: (id: string, filePath: string, source: 'external' | 'codex-generated') => this.registerAttachment(id, filePath, source), updateSessionName: (id: string, name: string) => this.mux.updateSessionName(id, name), + // Opt-in: the first prompt lands in the tab name, mux-sessions.json, every + // session:updated broadcast and /api/search, so it is a choice, not a default. + isAutoNameEnabled: async () => (await this.readSettings()).autoNameSessions === true, }; } diff --git a/src/web/session-listener-wiring.ts b/src/web/session-listener-wiring.ts index 2af8d7a0..6522c01a 100644 --- a/src/web/session-listener-wiring.ts +++ b/src/web/session-listener-wiring.ts @@ -29,7 +29,8 @@ import { getLifecycleLog } from '../session-lifecycle-log.js'; import { fileStreamManager } from '../file-stream-manager.js'; import { sessionWaits } from './session-wait-registry.js'; import { approvalInbox } from './approval-inbox.js'; -import { deriveAutoSessionName } from '../session-auto-name.js'; +import { composeAutoSessionName, deriveAutoSessionName } from '../session-auto-name.js'; +import { MAX_SESSION_NAME_LENGTH } from '../config/terminal-limits.js'; /** Stored listener references for session cleanup (prevents memory leaks) */ export interface SessionListenerRefs { @@ -86,6 +87,8 @@ interface SessionListenerDeps { getStore(): import('../state-store.js').StateStore; registerAttachment(sessionId: string, filePath: string, source: 'external' | 'codex-generated'): Promise<void>; updateSessionName(sessionId: string, name: string): boolean; + /** The synced `autoNameSessions` setting, read fresh so a flip applies to the next prompt. */ + isAutoNameEnabled(): Promise<boolean>; } /** @@ -455,13 +458,28 @@ export function createSessionListeners(session: Session, deps: SessionListenerDe }); }, - /** Assigns a bounded local title from the first real task prompt. */ + /** + * Names a placeholder tab after its first real prompt (`w3-case: fix the + * login redirect`), behind the synced `autoNameSessions` setting. The + * eligibility check comes first so the settings read costs nothing on the + * prompts of an already-named session; a prompt that yields no title (a + * slash command) leaves the session eligible for the next one. + */ promptSubmitted: (prompt: string) => { - const name = deriveAutoSessionName(prompt); - if (!name || !session.applyAutoName(name)) return; - deps.updateSessionName(session.id, session.name); - deps.persistSessionState(session); - deps.broadcast(SseEvent.SessionUpdated, deps.getSessionStateWithRespawn(session)); + if (session.nameSource !== 'placeholder') return; + const title = deriveAutoSessionName(prompt); + if (!title) return; + void deps + .isAutoNameEnabled() + .then((enabled) => { + if (!enabled) return; + const name = composeAutoSessionName(session.name, title, MAX_SESSION_NAME_LENGTH); + if (!session.applyAutoName(name)) return; + deps.updateSessionName(session.id, session.name); + deps.persistSessionState(session); + deps.broadcast(SseEvent.SessionUpdated, deps.getSessionStateWithRespawn(session)); + }) + .catch((err) => console.error(`[Session] auto-name failed for ${session.id}:`, err)); }, }; } diff --git a/test/mocks/mock-session.ts b/test/mocks/mock-session.ts index e32b8e68..4666afe0 100644 --- a/test/mocks/mock-session.ts +++ b/test/mocks/mock-session.ts @@ -68,6 +68,9 @@ export class MockSession extends EventEmitter { this.lastSubmitAt = Date.now(); } + /** Mirrors Session.trackUserInput (the send-key route feeds it around the write path). */ + trackUserInput(_data: string): void {} + private _muxName: string | null = null; constructor(id: string = 'mock-session-id') { diff --git a/test/routes/session-name-routes.test.ts b/test/routes/session-name-routes.test.ts new file mode 100644 index 00000000..ba161759 --- /dev/null +++ b/test/routes/session-name-routes.test.ts @@ -0,0 +1,60 @@ +/** + * @fileoverview PUT /api/sessions/:id/name hands the name to the user (#376). + * + * A rename flips `nameSource` to `manual`, persists it and broadcasts it, so + * auto-naming can never overwrite a name a person chose, on this server or + * on the one that restores the session after a restart. + * + * Uses app.inject() — no real HTTP ports needed. + */ + +import { describe, it, expect, beforeAll, afterAll, vi } from 'vitest'; +import { registerSessionRoutes } from '../../src/web/routes/session-routes.js'; +import { createRouteTestHarness, type RouteTestHarness } from './_route-test-utils.js'; +import { Session } from '../../src/session.js'; +import { SseEvent } from '../../src/web/sse-events.js'; + +describe('PUT /api/sessions/:id/name', () => { + let harness: RouteTestHarness; + let session: Session; + const updateSessionName = vi.fn(() => true); + + beforeAll(async () => { + harness = await createRouteTestHarness(registerSessionRoutes); + // A REAL session, since the ownership flag lives on the class, not the mock. + session = new Session({ id: 'name-route-test', workingDir: '/tmp', name: 'w1-demo' }); + harness.ctx.sessions.set(session.id, session as never); + (harness.ctx.mux as Record<string, unknown>).updateSessionName = updateSessionName; + }); + + afterAll(async () => { + await harness.app.close(); + }); + + it('flips a placeholder to manual, then persists and broadcasts the ownership', async () => { + expect(session.nameSource).toBe('placeholder'); + + const res = await harness.app.inject({ + method: 'PUT', + url: `/api/sessions/${session.id}/name`, + payload: { name: 'my window' }, + }); + + expect(res.statusCode).toBe(200); + // The harness registers the bare route; the {success,data} envelope is a server-level hook. + expect(res.json()).toMatchObject({ name: 'my window' }); + expect(session.name).toBe('my window'); + expect(session.nameSource).toBe('manual'); + expect(session.applyAutoName('w1-demo: fix it')).toBe(false); + expect(session.name).toBe('my window'); + + expect(updateSessionName).toHaveBeenCalledWith(session.id, 'my window'); + expect(harness.ctx.persistSessionState).toHaveBeenCalledWith(session); + expect(harness.ctx.broadcast).toHaveBeenCalledWith( + SseEvent.SessionUpdated, + expect.objectContaining({ id: session.id, name: 'my window', nameSource: 'manual' }) + ); + // What the restore path will read back: the persisted state carries the flag. + expect(session.toState().nameSource).toBe('manual'); + }); +}); diff --git a/test/session-auto-name.test.ts b/test/session-auto-name.test.ts new file mode 100644 index 00000000..7d997071 --- /dev/null +++ b/test/session-auto-name.test.ts @@ -0,0 +1,274 @@ +/** + * @fileoverview Auto-naming a session after its first prompt (#376). + * + * The tracker sits on the raw keystroke stream, so most of these pin the + * per-key rules that a review of the first cut found missing: a bare Esc ate + * the next prompt's first character, a wheel report mid-word dropped half the + * prompt, pasted newlines counted as Enter, and every prompt renamed the tab. + * + * Port: N/A (no server needed) + */ + +import { describe, it, expect, vi } from 'vitest'; +import { Session } from '../src/session.js'; +import { + SubmittedPromptTracker, + deriveAutoSessionName, + composeAutoSessionName, + isGeneratedSessionName, +} from '../src/session-auto-name.js'; + +describe('SubmittedPromptTracker', () => { + it('reports the draft on Enter across arbitrary chunks, honouring backspace', () => { + const tracker = new SubmittedPromptTracker(); + expect(tracker.feed('fix the')).toEqual([]); + expect(tracker.feed(' login bugs\x7f')).toEqual([]); + expect(tracker.feed('\r')).toEqual(['fix the login bug']); + expect(tracker.feed('\r')).toEqual([]); + expect(tracker.feed('修复登录跳转\x08问题\r')).toEqual(['修复登录跳问题']); + }); + + it('treats a bare Esc as the Esc key, not the start of a sequence', () => { + const tracker = new SubmittedPromptTracker(); + tracker.feed('\x1b'); + expect(tracker.feed('fix the login bug\r')).toEqual(['fix the login bug']); + tracker.feed('\x1b'); + expect(tracker.feed('修复登录\r')).toEqual(['修复登录']); + // Esc then digits and punctuation used to grow the escape buffer without bound. + tracker.feed('\x1b'); + expect(tracker.feed('12345, ok?\r')).toEqual(['12345, ok?']); + // A double Esc is two Esc keys, each its own write (in ONE chunk, `ESC s` + // is Alt+s by the terminal's own encoding and stays swallowed). + tracker.feed('\x1b'); + tracker.feed('\x1b'); + expect(tracker.feed('still here\r')).toEqual(['still here']); + }); + + it('swallows Alt chords and turns Alt+Enter into a newline in the draft', () => { + const tracker = new SubmittedPromptTracker(); + expect(tracker.feed('fix\x1bb the\x1b\rbug\r')).toEqual(['fix the bug']); + }); + + it('ignores cursor keys, mouse and focus reports, Shift+Tab and Tab', () => { + const tracker = new SubmittedPromptTracker(); + expect(tracker.feed('fix the \x1b[<64;10;5M\x1b[<65;10;5mlogin bug\r')).toEqual(['fix the login bug']); + expect(tracker.feed('look at @src/ses\tsion.ts and fix it\r')).toEqual(['look at @src/session.ts and fix it']); + expect(tracker.feed('typo\x1b[D\x1b[C\x1b[H\x1b[F\x1b[3~\x1b[Z\x1b[I\x1b[O\x1bOC fixed\r')).toEqual(['typo fixed']); + expect(tracker.feed('mod\x1b[1;5D\x1b[1;2Cifiers\r')).toEqual(['modifiers']); + }); + + it('taints the draft on history recall so Enter submits nothing rather than a fragment', () => { + const tracker = new SubmittedPromptTracker(); + expect(tracker.feed('old text\x1b[A and more\r')).toEqual([]); + expect(tracker.feed('\x1bOB\r')).toEqual([]); + expect(tracker.feed('\x1b[1;5A\r')).toEqual([]); + expect(tracker.feed('\x10x\r')).toEqual([]); + expect(tracker.feed('\x12search\r')).toEqual([]); + expect(tracker.feed('fresh prompt\r')).toEqual(['fresh prompt']); + // Ctrl+C empties the composer, which also clears the taint. + expect(tracker.feed('stale\x1b[A\x03typed after\r')).toEqual(['typed after']); + }); + + it('keeps bracketed-paste newlines inside the draft', () => { + const tracker = new SubmittedPromptTracker(); + expect(tracker.feed('\x1b[200~line one\nline two\r\nline three\x1b[201~ plus typed\r')).toEqual([ + 'line one line two line three plus typed', + ]); + // A paste split across chunks stays a paste. + tracker.feed('\x1b[200~first\r'); + expect(tracker.feed('second\x1b[201~\r')).toEqual(['first second']); + }); + + it('joins a Shift+Enter / Ctrl+J newline with a space', () => { + const tracker = new SubmittedPromptTracker(); + tracker.feed('Fix the login bug'); + tracker.feed('\n'); + expect(tracker.feed('Also add tests.\r')).toEqual(['Fix the login bug Also add tests.']); + }); + + it('mirrors Ctrl+W, Ctrl+U and Ctrl+C', () => { + const tracker = new SubmittedPromptTracker(); + expect(tracker.feed('fix the bugs\x17bug\r')).toEqual(['fix the bug']); + expect(tracker.feed('discarded\x15kept\r')).toEqual(['kept']); + expect(tracker.feed('discarded\x03kept\r')).toEqual(['kept']); + }); + + it('keeps the HEAD of an over-long draft', () => { + const tracker = new SubmittedPromptTracker(); + const [prompt] = tracker.feed(`${'a'.repeat(9000)}\r`); + expect(prompt).toHaveLength(8192); + // Backspaces past the cap consume the overflow before the kept text. + const [again] = tracker.feed(`${'b'.repeat(8200)}${'\x7f'.repeat(10)}\r`); + expect(again).toHaveLength(8190); + }); + + it('abandons a malformed escape without eating the text, and taints on an over-long one', () => { + const tracker = new SubmittedPromptTracker(); + expect(tracker.feed('\x1b[修复\r')).toEqual(['修复']); + expect(tracker.feed('\x1b]0;window title\x07hello\r')).toEqual(['hello']); + // Nothing a terminal sends runs past 64 bytes; the tail is garbage, not a title. + expect(tracker.feed(`\x1b]${'x'.repeat(80)}after\r`)).toEqual([]); + expect(tracker.feed('next prompt\r')).toEqual(['next prompt']); + }); + + it('resumes a CSI split across chunks', () => { + const tracker = new SubmittedPromptTracker(); + tracker.feed('abc\x1b['); + expect(tracker.feed('Ddef\r')).toEqual(['abcdef']); + }); +}); + +describe('deriveAutoSessionName', () => { + it('takes the first sentence, drops the full stop, and bounds the length', () => { + expect(deriveAutoSessionName('Fix the login bug. Also add tests.')).toBe('Fix the login bug'); + expect(deriveAutoSessionName(' 修复登录跳转问题。\n不要改数据库')).toBe('修复登录跳转问题'); + expect(deriveAutoSessionName('Why does this crash? It worked before')).toBe('Why does this crash?'); + expect(deriveAutoSessionName('Run v2.0 tests. Then deploy')).toBe('Run v2.0 tests'); + expect(Array.from(deriveAutoSessionName('a'.repeat(200)) ?? '')).toHaveLength(72); + const cut = deriveAutoSessionName('word '.repeat(40).trim()) ?? ''; + expect(cut.endsWith('…')).toBe(true); + expect(cut).toMatch(/^(word )+word…$/); + }); + + it('does not cut on an abbreviation early in the prompt', () => { + expect(deriveAutoSessionName('e.g. fix this now')).toBe('e.g. fix this now'); + expect(deriveAutoSessionName('Ok. Fix the login bug')).toBe('Ok. Fix the login bug'); + }); + + it('returns null for commands and empties, but not for paths', () => { + expect(deriveAutoSessionName('/clear')).toBeNull(); + expect(deriveAutoSessionName('/model opus')).toBeNull(); + expect(deriveAutoSessionName('/ralph-loop:ralph-loop')).toBeNull(); + expect(deriveAutoSessionName('! npm test')).toBeNull(); + expect(deriveAutoSessionName(' ')).toBeNull(); + expect(deriveAutoSessionName('/home/me/notes.txt what is this')).toBe('/home/me/notes.txt what is this'); + }); + + it('strips control bytes and ANSI before the title is persisted', () => { + expect(deriveAutoSessionName('\x1b[31m整理项目文档\x1b[0m')).toBe('整理项目文档'); + expect(deriveAutoSessionName('a\x00b\tc')).toBe('a b c'); + }); +}); + +describe('composeAutoSessionName', () => { + it('keeps the placeholder as a prefix so the case and the counter survive', () => { + expect(composeAutoSessionName('w3-myapp', 'fix the login bug')).toBe('w3-myapp: fix the login bug'); + expect(composeAutoSessionName('', 'fix the login bug')).toBe('fix the login bug'); + }); + + it('honours the rename cap in UTF-16 units', () => { + const name = composeAutoSessionName('w3-myapp', '😀'.repeat(100), 40); + expect(name.length).toBeLessThanOrEqual(40); + expect(name.startsWith('w3-myapp: ')).toBe(true); + expect(name.endsWith('…')).toBe(true); + expect(composeAutoSessionName('x'.repeat(127), 'title', 128)).toBe('x'.repeat(127)); + }); + + it('recognises only the generated w/s + number + case form', () => { + expect(isGeneratedSessionName('w12-my_case-2')).toBe(true); + expect(isGeneratedSessionName('s1-shell')).toBe(true); + expect(isGeneratedSessionName('w1-case: fix it')).toBe(false); + expect(isGeneratedSessionName('alpha')).toBe(false); + }); +}); + +describe('Session name ownership', () => { + it('infers placeholder vs manual from the name and persists the source', () => { + const placeholder = new Session({ workingDir: '/tmp', name: 'w1-demo' }); + expect(placeholder.nameSource).toBe('placeholder'); + expect(placeholder.toState().nameSource).toBe('placeholder'); + expect(new Session({ workingDir: '/tmp' }).nameSource).toBe('placeholder'); + expect(new Session({ workingDir: '/tmp', name: 'my window' }).nameSource).toBe('manual'); + expect(new Session({ workingDir: '/tmp', name: 'w1-demo: fix it' }).nameSource).toBe('manual'); + // The boot restore passes the persisted source, which outranks the inference. + const recovered = new Session({ workingDir: '/tmp', name: 'w1-demo: fix it', nameSource: 'auto' }); + expect(recovered.nameSource).toBe('auto'); + expect(recovered.applyAutoName('w1-demo: other')).toBe(false); + }); + + it('names once: the first prompt takes it, later prompts and renames do not', () => { + const session = new Session({ workingDir: '/tmp', name: 'w1-demo' }); + expect(session.applyAutoName('w1-demo: fix the login bug')).toBe(true); + expect(session.name).toBe('w1-demo: fix the login bug'); + expect(session.nameSource).toBe('auto'); + expect(session.applyAutoName('w1-demo: 1')).toBe(false); + expect(session.name).toBe('w1-demo: fix the login bug'); + + session.name = 'mine'; + expect(session.nameSource).toBe('manual'); + expect(session.applyAutoName('other')).toBe(false); + expect(session.name).toBe('mine'); + }); + + it('consumes the first prompt even when the composed name is unchanged', () => { + const session = new Session({ workingDir: '/tmp', name: 'w1-demo' }); + expect(session.applyAutoName('w1-demo')).toBe(false); + expect(session.nameSource).toBe('auto'); + }); +}); + +describe('Session promptSubmitted', () => { + function withFakePty(session: Session): ReturnType<typeof vi.fn> { + const write = vi.fn(); + (session as unknown as { ptyProcess: { write: typeof write } }).ptyProcess = { write }; + return write; + } + + it('emits for user input only, after the bytes reached the PTY', () => { + const session = new Session({ workingDir: '/tmp', name: 'w1-demo' }); + const prompts: string[] = []; + session.on('promptSubmitted', (p: string) => prompts.push(p)); + + // No PTY yet: the write fails and nothing is reported. + expect(session.write('lost\r', { fromUser: true })).toBe(false); + expect(prompts).toEqual([]); + + const write = withFakePty(session); + expect(session.write('Read @ralph_prompt.md and follow the instructions.\r')).toBe(true); + expect(prompts).toEqual([]); + expect(session.write('fix the ', { fromUser: true })).toBe(true); + expect(session.write('login bug\r', { fromUser: true })).toBe(true); + expect(prompts).toEqual(['fix the login bug']); + expect(write).toHaveBeenCalledTimes(3); + // The pane's last-Enter stamp is kept for EVERY write, user or not. + expect(session.lastSubmitAt).toBeGreaterThan(0); + }); + + it('never feeds the tracker for a shell session', () => { + const session = new Session({ workingDir: '/tmp', name: 's1-demo', mode: 'shell' }); + const prompts: string[] = []; + session.on('promptSubmitted', (p: string) => prompts.push(p)); + withFakePty(session); + expect(session.write('ls -la\r', { fromUser: true })).toBe(true); + session.trackUserInput('cd src\r'); + expect(prompts).toEqual([]); + }); + + it('feeds the send-key line feed so a two-line prompt keeps its separator', () => { + const session = new Session({ workingDir: '/tmp', name: 'w1-demo' }); + const prompts: string[] = []; + session.on('promptSubmitted', (p: string) => prompts.push(p)); + withFakePty(session); + session.write('Fix the login bug', { fromUser: true }); + session.trackUserInput('\n'); + session.write('Also add tests.\r', { fromUser: true }); + expect(prompts).toEqual(['Fix the login bug Also add tests.']); + }); + + it('reports through writeViaMux only when the mux accepted the input', async () => { + const session = new Session({ workingDir: '/tmp', name: 'w1-demo' }); + const prompts: string[] = []; + session.on('promptSubmitted', (p: string) => prompts.push(p)); + const sendInput = vi.fn(async () => false); + (session as unknown as { _mux: unknown; _muxSession: unknown })._mux = { sendInput }; + (session as unknown as { _mux: unknown; _muxSession: unknown })._muxSession = { sessionId: session.id }; + + expect(await session.writeViaMux('dropped\r', { fromUser: true })).toBe(false); + expect(prompts).toEqual([]); + sendInput.mockResolvedValue(true); + expect(await session.writeViaMux('delivered\r', { fromUser: true })).toBe(true); + expect(prompts).toEqual(['delivered']); + expect(await session.writeViaMux('/clear\r')).toBe(true); + expect(prompts).toEqual(['delivered']); + }); +}); diff --git a/test/session-listener-wiring.test.ts b/test/session-listener-wiring.test.ts index 5ffd56e0..1173fecb 100644 --- a/test/session-listener-wiring.test.ts +++ b/test/session-listener-wiring.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it, vi } from 'vitest'; import { Session } from '../src/session.js'; import { createSessionListeners } from '../src/web/session-listener-wiring.js'; +import { SseEvent } from '../src/web/sse-events.js'; describe('session listener wiring', () => { it('forwards the attachment request source through registerAttachment', async () => { @@ -21,30 +22,69 @@ describe('session listener wiring', () => { expect(registerAttachment).toHaveBeenNthCalledWith(2, 'wiring-attach-source-test', '/tmp/report.pdf', 'external'); }); - it('renames an eligible session when its first prompt is submitted', () => { - const session = new Session({ id: 'wiring-auto-name-test', workingDir: '/tmp', name: 'w1-demo' }); - const updateSessionName = vi.fn(() => true); - const persistSessionState = vi.fn(); - const broadcast = vi.fn(); - const getSessionStateWithRespawn = vi.fn(() => session.toState()); + /** The listener reads the setting asynchronously; let its promise chain settle. */ + const flush = () => new Promise((resolve) => setTimeout(resolve, 5)); + + function autoNameDeps(session: Session, enabled: boolean) { const deps = { - updateSessionName, - persistSessionState, - broadcast, - getSessionStateWithRespawn, - } as unknown as Parameters<typeof createSessionListeners>[1]; + updateSessionName: vi.fn(() => true), + persistSessionState: vi.fn(), + broadcast: vi.fn(), + getSessionStateWithRespawn: vi.fn(() => session.toState()), + isAutoNameEnabled: vi.fn(async () => enabled), + }; + return { + deps, + refs: createSessionListeners(session, deps as unknown as Parameters<typeof createSessionListeners>[1]), + }; + } + + it('names a placeholder tab after its first real prompt, in the prefix form, once', async () => { + const session = new Session({ id: 'wiring-auto-name-test', workingDir: '/tmp', name: 'w1-demo' }); + const { deps, refs } = autoNameDeps(session, true); + + // A slash command yields no title and leaves the session eligible; the + // setting is not even read for it. + refs.promptSubmitted('/clear'); + await flush(); + expect(deps.isAutoNameEnabled).not.toHaveBeenCalled(); + expect(session.name).toBe('w1-demo'); - const refs = createSessionListeners(session, deps); refs.promptSubmitted('整理登录模块并补充测试'); + await flush(); + expect(session.name).toBe('w1-demo: 整理登录模块并补充测试'); + expect(session.nameSource).toBe('auto'); + expect(deps.updateSessionName).toHaveBeenCalledWith('wiring-auto-name-test', 'w1-demo: 整理登录模块并补充测试'); + expect(deps.persistSessionState).toHaveBeenCalledWith(session); + expect(deps.broadcast).toHaveBeenCalledWith( + SseEvent.SessionUpdated, + expect.objectContaining({ name: 'w1-demo: 整理登录模块并补充测试', nameSource: 'auto' }) + ); - expect(session.name).toBe('整理登录模块并补充测试'); - expect(updateSessionName).toHaveBeenCalledWith('wiring-auto-name-test', '整理登录模块并补充测试'); - expect(persistSessionState).toHaveBeenCalledWith(session); - expect(broadcast).toHaveBeenCalled(); + // The second prompt never reaches the setting: the tab is named. + refs.promptSubmitted('1'); + await flush(); + expect(deps.isAutoNameEnabled).toHaveBeenCalledTimes(1); + expect(session.name).toBe('w1-demo: 整理登录模块并补充测试'); + }); + + it('leaves the tab alone while the setting is off, and never touches a manual name', async () => { + const session = new Session({ id: 'wiring-auto-name-off', workingDir: '/tmp', name: 'w1-demo' }); + const { deps, refs } = autoNameDeps(session, false); + + refs.promptSubmitted('fix the login bug'); + await flush(); + expect(deps.isAutoNameEnabled).toHaveBeenCalledTimes(1); + expect(session.name).toBe('w1-demo'); + // Still a placeholder: flipping the setting on names the NEXT prompt. + expect(session.nameSource).toBe('placeholder'); + expect(deps.updateSessionName).not.toHaveBeenCalled(); session.name = '人工命名'; refs.promptSubmitted('新的任务不能覆盖人工命名'); + await flush(); + expect(deps.isAutoNameEnabled).toHaveBeenCalledTimes(1); expect(session.name).toBe('人工命名'); - expect(updateSessionName).toHaveBeenCalledTimes(1); + expect(deps.persistSessionState).not.toHaveBeenCalled(); }); }); diff --git a/test/session-submit-anchor.test.ts b/test/session-submit-anchor.test.ts index 5c75564c..a1298fb9 100644 --- a/test/session-submit-anchor.test.ts +++ b/test/session-submit-anchor.test.ts @@ -12,7 +12,6 @@ import { describe, it, expect } from 'vitest'; import { Session } from '../src/session.js'; -import { deriveAutoSessionName, SubmittedPromptTracker } from '../src/session-auto-name.js'; describe('session submit anchor', () => { it('records the pane Enter and carries it into persisted state', () => { @@ -54,31 +53,3 @@ describe('session submit anchor', () => { expect(recovered.lastSubmitAt).toBe(0); }); }); - -describe('automatic session names', () => { - it('builds a bounded title from the first sentence without exposing controls', () => { - expect(deriveAutoSessionName(' 修复登录跳转问题。\n不要改数据库')).toBe('修复登录跳转问题。'); - expect(deriveAutoSessionName('/clear')).toBeNull(); - expect(deriveAutoSessionName('\x1b[31m整理项目文档\x1b[0m')).toBe('整理项目文档'); - expect(Array.from(deriveAutoSessionName('a'.repeat(200)) ?? '')).toHaveLength(72); - }); - - it('tracks chunked typing, backspace, and Enter without treating arrows as prompt text', () => { - const tracker = new SubmittedPromptTracker(); - expect(tracker.feed('修复登')).toEqual([]); - expect(tracker.feed('录跳转\x7f问题\r')).toEqual(['修复登录跳问题']); - expect(tracker.feed('旧内容\x1b[A新内容\r')).toEqual(['新内容']); - }); - - it('keeps manual names protected while generated names remain eligible', () => { - const generated = new Session({ workingDir: '/tmp', name: 'w1-demo' }); - expect(generated.nameSource).toBe('auto'); - expect(generated.applyAutoName('修复登录')).toBe(true); - expect(generated.applyAutoName('继续重命名')).toBe(true); - - const manual = new Session({ workingDir: '/tmp', name: '我的工作窗口' }); - expect(manual.nameSource).toBe('manual'); - expect(manual.applyAutoName('不应覆盖')).toBe(false); - expect(manual.name).toBe('我的工作窗口'); - }); -}); From c9515b1d4c6d2469179506b78bb9d4b3956582b7 Mon Sep 17 00:00:00 2001 From: Michael Grundberg <michael.grundberg@irisity.com> Date: Tue, 15 Sep 2026 12:56:09 +0200 Subject: [PATCH 06/28] fix(terminal): keep the output a pane capture could not contain Live terminal events are queued while a buffer load runs, and the load discards that queue when it ends. That is right when the loaded buffer is the server's accumulated byte history. The route appends to that history right up to the moment it serializes the response, so a queued event already appears in it and replaying it would duplicate output, most visibly Ink's cursor-up redraws. A tmux pane capture is a photograph, current only as of the instant `capture-pane` ran. Output printed afterwards was queued and then dropped, and nothing scheduled a re-fetch to recover it: `_onSessionNeedsRefresh` is wired only to the 128KB overflow path. The CLI's next partial redraw then landed on a frame the terminal never received. How much went missing depended on which capture the route served. A `?full=1` load returns the capture alone, with no history in front of it, so it lost everything from the capture to the end of the chunked write. A `?tail=` load returns history, a clear, and then the capture, and the route reads that history after the capture, so it lost everything from the response to the end of that write. The chunked write dominates either way. An agent CLI hides the loss on its next full redraw; a shell session does not, because its output is linear and nothing repaints it. Queue entries now carry their arrival time, and `_finishBufferLoad` takes a `since` cutoff, so a capture load replays exactly the tail that arrived after the response headers. The earlier events stay dropped, because a payload that carries history does hold those. All four paths that fetch a terminal buffer and write it now decide this the same way, through one `_bufferLoadFinishOpts` helper, so they cannot drift apart: `selectSession`, `_onSessionNeedsRefresh`, `_onSessionClearTerminal` and `_maybeRefetchFullHistory`. The second of those is the one that stings. It exists to restore output the client already dropped once under backpressure, and it was dropping more output while performing that recovery. The cache-hit write inside `selectSession` stays on discard deliberately: it runs before the fetch, so its queue holds only events the capture that follows already contains. Two further things had to change for that tail to still exist when the load ends, and a browser test is what found both. `chunkedTerminalWrite` is what ends the load for every non-empty buffer, so the flush policy travels to its own finish calls; the call in `selectSession` runs only when the write was skipped. `_beginBufferLoad` no longer empties the queue when one load re-enters it, which it does on every write, because that reset discarded the whole fetch window before anything could replay it. The response already distinguishes the sources. `source` reads `mux-visible` or `mux-full-history` for a capture and `history` for the byte stream. Follows #395, #396 and #397, which fixed the ways the replayed frame itself could disagree with the terminal. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- ...y-output-that-arrived-after-the-capture.md | 36 ++++ config/test-suites.ts | 1 + src/web/public/app.js | 71 ++++++- src/web/public/terminal-ui.js | 57 ++++-- test/capture-load-window.browser.test.ts | 177 ++++++++++++++++++ test/terminal-buffer-flush.test.ts | 97 +++++++++- 6 files changed, 413 insertions(+), 26 deletions(-) create mode 100644 .changeset/fix-replay-output-that-arrived-after-the-capture.md create mode 100644 test/capture-load-window.browser.test.ts diff --git a/.changeset/fix-replay-output-that-arrived-after-the-capture.md b/.changeset/fix-replay-output-that-arrived-after-the-capture.md new file mode 100644 index 00000000..03e09189 --- /dev/null +++ b/.changeset/fix-replay-output-that-arrived-after-the-capture.md @@ -0,0 +1,36 @@ +--- +"aicodeman": patch +--- + +fix(terminal): keep the output a pane capture could not contain + +Live terminal events are queued while a buffer load runs, and the load discards +that queue when it ends. That is right when the loaded buffer is the server's +accumulated byte history: the route appends to that history right up to the +moment it serializes the response, so the queued events already appear in it and +replaying them would duplicate output. + +A tmux pane capture is a photograph, current only as of the instant +`capture-pane` ran. Output printed afterwards was queued and then dropped, with +nothing scheduling a re-fetch, and the CLI's next partial redraw landed on a +frame the terminal never received. A `?full=1` load returns the capture alone, +so it lost everything from the capture to the end of the chunked write. A +`?tail=` load carries the byte history in front of the capture, so it lost +everything from the response to the end of that write. A shell session shows +this most plainly, because its output is linear and nothing repaints it. + +Queue entries now carry their arrival time, and `_finishBufferLoad` takes a +`since` cutoff so a capture load replays exactly the tail that arrived after the +response headers. All four paths that fetch a terminal buffer and write it use +the same rule, through one shared `_bufferLoadFinishOpts` helper: selecting a +session, the backpressure refresh, the clear-terminal reload, and the +full-history re-pull. The backpressure refresh matters most, because it exists +to restore output the client already dropped once and could drop more while +doing it. + +Two things had to change for that tail to still exist when the load ends. +`chunkedTerminalWrite` is what ends the load for any non-empty buffer, so it +takes the flush policy and applies it at its own finish sites. +`_beginBufferLoad` no longer empties the queue when the same load re-enters it, +which it does on every write, because that reset discarded the fetch window +before anything could replay it. diff --git a/config/test-suites.ts b/config/test-suites.ts index 02cc154b..233e0e7d 100644 --- a/config/test-suites.ts +++ b/config/test-suites.ts @@ -27,6 +27,7 @@ export const BROWSER_TEST_GLOBS = [ 'test/webgl-fallback.test.ts', 'test/terminal-copy-shortcut.test.ts', 'test/terminal-keycode229-recovery.browser.test.ts', + 'test/capture-load-window.browser.test.ts', 'test/codex-predictive-echo.test.ts', // also needs a real codex binary ]; diff --git a/src/web/public/app.js b/src/web/public/app.js index f91f748b..ab303da5 100644 --- a/src/web/public/app.js +++ b/src/web/public/app.js @@ -1906,6 +1906,30 @@ class CodemanApp { this._onSessionClearTerminal(data); } + /** + * How a buffer load that just fetched `payload` must end. + * + * A tmux pane capture is a point-in-time frame, so nothing that reached the + * browser after the response headers can already be in it. Such a load + * replays exactly that tail; discarding it drops the CLI's output for the + * rest of the load window, and its next partial redraw then lands on a frame + * the terminal never received. A payload built from the server's accumulated + * byte history needs the opposite: that history is current up to the + * response, so replaying the queue on top of it would duplicate output. + * + * `headersReceivedAt` is the caller's own `performance.now()` reading from + * the moment the response arrived, compared only against other client-side + * readings, so there is no clock skew to worry about. + * + * @param {{source?: string}} payload - The parsed `data` of a terminal response. + * @param {number} headersReceivedAt - When that response reached this client. + * @returns {{flushQueued: boolean, since: number}} Options for `_finishBufferLoad`. + */ + _bufferLoadFinishOpts(payload, headersReceivedAt) { + const capturedFromMux = payload?.source === 'mux-visible' || payload?.source === 'mux-full-history'; + return { flushQueued: capturedFromMux, since: headersReceivedAt }; + } + _onSessionTerminal(data) { if (data.id === this.activeSessionId) { if (data.data.length > 32768) _crashDiag.log(`TERMINAL: ${(data.data.length/1024).toFixed(0)}KB`); @@ -1915,7 +1939,7 @@ class CodemanApp { // jump over the cap. Dropped data is recovered from the canonical buffer. const queued = (this.pendingWrites?.reduce((s, w) => s + w.length, 0) || 0) + (this.flickerFilterBuffer?.length || 0) - + (this._loadBufferQueue?.reduce((s, w) => s + w.length, 0) || 0) + + (this._loadBufferQueue?.reduce((s, w) => s + w.data.length, 0) || 0) + (this._terminalWriteInFlightBytes || 0); if (queued + data.data.length > 131072) { // 128KB — drop to prevent accumulation // Schedule a self-recovery once the @@ -2498,9 +2522,11 @@ class CodemanApp { ? `/api/sessions/${sessionId}/terminal?full=1` : `/api/sessions/${sessionId}/terminal?tail=${TERMINAL_TAIL_SIZE}` ); + let headersReceivedAt = performance.now(); let data = (await res.json())?.data ?? {}; if (useFullHistory && data.terminalBuffer && this._replayWouldShrinkBuffer(data.terminalBuffer)) { res = await fetch(`/api/sessions/${sessionId}/terminal?tail=${TERMINAL_TAIL_SIZE}`); + headersReceivedAt = performance.now(); data = (await res.json())?.data ?? {}; } // Bail on a tab switch mid-fetch: writing here would paint this session's @@ -2516,7 +2542,12 @@ class CodemanApp { const linesFromBottom = before ? Math.max(0, (before.baseY || 0) - (before.viewportY || 0)) : 0; this.terminal.clear(); this.terminal.reset(); - await this.chunkedTerminalWrite(data.terminalBuffer); + await this.chunkedTerminalWrite( + data.terminalBuffer, + TERMINAL_CHUNK_SIZE, + undefined, + this._bufferLoadFinishOpts(data, headersReceivedAt) + ); // A tail fetch can be partial, and the banner would otherwise keep // describing the pre-refresh buffer (#258). this._setHistoryTruncation(sessionId, data); @@ -2552,6 +2583,7 @@ class CodemanApp { // Fetch buffer, clear terminal, write buffer, resize (no Ctrl+L needed) try { const res = await fetch(`/api/sessions/${data.id}/terminal`); + const headersReceivedAt = performance.now(); const termData = (await res.json())?.data ?? {}; this.terminal.clear(); @@ -2561,7 +2593,12 @@ class CodemanApp { // (markers don't help here - this is a static buffer reload, not live Ink redraws) const cleanBuffer = termData.terminalBuffer.replace(DEC_SYNC_STRIP_RE, ''); // Use chunked write to avoid UI freeze with large buffers (can be 1-2MB) - await this.chunkedTerminalWrite(cleanBuffer); + await this.chunkedTerminalWrite( + cleanBuffer, + TERMINAL_CHUNK_SIZE, + undefined, + this._bufferLoadFinishOpts(termData, headersReceivedAt) + ); } // Fire-and-forget resize — don't block on it @@ -5782,7 +5819,12 @@ class CodemanApp { parsedAt, bufferLength: parsedBufferLength, completed, - } = await this.chunkedTerminalWrite(buffer, TERMINAL_CHUNK_SIZE, sessionId); + } = await this.chunkedTerminalWrite( + buffer, + TERMINAL_CHUNK_SIZE, + sessionId, + this._bufferLoadFinishOpts(payload, headersReceivedAt) + ); timing.resetAndParseMs = parsedAt - replayStartedAt; if (!completed || this.activeSessionId !== sessionId) return; // Keep shell tab restores bounded too. A user-triggered full-history pull @@ -6241,6 +6283,15 @@ class CodemanApp { } const data = (await res.json())?.data ?? {}; const bodyParsedAt = performance.now(); + // How this load must end, decided here because `chunkedTerminalWrite` is + // what actually ends it for a non-empty buffer. A tmux pane capture is a + // point-in-time frame, so nothing that reached the browser after the + // response headers can already be in it. Replay exactly that tail; + // discarding it drops the CLI's output for the rest of the load window, + // and its next partial redraw then lands on a frame the terminal never + // received. `since` keeps the pre-capture events dropped, because the + // capture does hold those and replaying them would duplicate output. + const finishOpts = this._bufferLoadFinishOpts(data, headersReceivedAt); _crashDiag.log(`FETCH_DONE: ${data.terminalBuffer ? (data.terminalBuffer.length/1024).toFixed(0) + 'KB' : 'empty'} truncated=${data.truncated}`); let freshResetAndParseMs = 0; @@ -6267,7 +6318,8 @@ class CodemanApp { const { parsedAt: freshParsedAt } = await this.chunkedTerminalWrite( data.terminalBuffer, TERMINAL_CHUNK_SIZE, - bufferLoadOwner + bufferLoadOwner, + finishOpts ); freshResetAndParseMs = freshParsedAt - replayStartedAt; if (this._isStaleSelect(selectGen)) { @@ -6317,7 +6369,14 @@ class CodemanApp { // COD-144: when the load painted nothing, FLUSH the queued events instead of // discarding — a new session's prompt arrives only as a queued SSE event. if (this._isLoadingBuffer) { - this._finishBufferLoad(bufferLoadOwner, { flushQueued: bufferWasEmpty }); + // Only reached when the write was skipped. COD-144 lives here: a new + // session's first prompt exists only as a queued event that predates the + // response, so an empty paint replays its queue WHOLE rather than from + // the header timestamp. + this._finishBufferLoad( + bufferLoadOwner, + bufferWasEmpty ? { flushQueued: true, since: 0 } : finishOpts + ); } // Drop the guard so user input clears state normally this._restoringFlushedState = false; diff --git a/src/web/public/terminal-ui.js b/src/web/public/terminal-ui.js index 85c13350..1a265d61 100644 --- a/src/web/public/terminal-ui.js +++ b/src/web/public/terminal-ui.js @@ -3267,7 +3267,11 @@ Object.assign(CodemanApp.prototype, { // to prevent interleaving historical buffer data with live SSE data. // This is critical: interleaving causes cursor position chaos with Ink redraws. if (this._isLoadingBuffer) { - if (this._loadBufferQueue) this._loadBufferQueue.push(data); + // Each entry records when it arrived. A flush of a tmux-capture load + // replays only what arrived after the capture; without the timestamp it + // would have to replay the whole queue, duplicating the events the + // capture already contains. See _finishBufferLoad's `since`. + if (this._loadBufferQueue) this._loadBufferQueue.push({ at: performance.now(), data }); return; } @@ -3747,9 +3751,14 @@ Object.assign(CodemanApp.prototype, { * and a tick-Worker so progress continues on occluded / idle-throttled tabs. * @param {string} buffer - The full terminal buffer to write * @param {number} chunkSize - Size of each chunk (default 32KB) + * @param {string} [loadOwner] - Load token to finish under + * @param {{ flushQueued?: boolean, since?: number }} [finishOpts] - Passed to + * `_finishBufferLoad`. This method ends the load for every non-empty buffer, + * so a caller that wants the queue replayed has to say so HERE; the call in + * `selectSession` only runs when the write was skipped entirely. * @returns {Promise<{parsedAt: number, bufferLength: number, completed: boolean}>} Parse marker snapshot */ - chunkedTerminalWrite(buffer, chunkSize = TERMINAL_CHUNK_SIZE, loadOwner) { + chunkedTerminalWrite(buffer, chunkSize = TERMINAL_CHUNK_SIZE, loadOwner, finishOpts) { // Generation counter: if a newer chunkedTerminalWrite starts (tab switch), // older writes abort instead of continuing to push stale data into the terminal. const writeGen = ++this._chunkedWriteGen; @@ -3762,7 +3771,7 @@ Object.assign(CodemanApp.prototype, { completed, }); if (!buffer || buffer.length === 0) { - this._finishBufferLoad(bufferLoadOwner); + this._finishBufferLoad(bufferLoadOwner, finishOpts); resolve(parseSnapshot()); return; } @@ -3776,7 +3785,7 @@ Object.assign(CodemanApp.prototype, { this.terminal.write(cleanBuffer, () => resolve(parseSnapshot())); // The write is now ordered in xterm's queue. Release live output before // parsing completes; subsequent writes stay behind it without being lost. - this._finishBufferLoad(bufferLoadOwner); + this._finishBufferLoad(bufferLoadOwner, finishOpts); return; } @@ -3807,7 +3816,7 @@ Object.assign(CodemanApp.prototype, { ); resolve(result); }); - this._finishBufferLoad(bufferLoadOwner); + this._finishBufferLoad(bufferLoadOwner, finishOpts); return; } @@ -3826,10 +3835,20 @@ Object.assign(CodemanApp.prototype, { * Called when chunkedTerminalWrite finishes (or is skipped for empty buffers). * * By default queued SSE events are DISCARDED, not flushed. For an established - * session the loaded buffer from the API is the source of truth up to the - * response timestamp; SSE events queued during the fetch+write overlap already - * appear in that buffer, so flushing them writes duplicate data (especially Ink - * cursor-up redraws), corrupting the terminal display. + * session whose buffer came from the server's accumulated byte history, that + * history is the source of truth up to the response timestamp; SSE events + * queued during the fetch+write overlap already appear in it, so flushing + * them writes duplicate data (especially Ink cursor-up redraws), corrupting + * the terminal display. + * + * A tmux PANE CAPTURE is the exception, and the reason `since` exists. A + * capture is a point-in-time frame taken part-way through the fetch, so it is + * the source of truth only up to CAPTURE time — not up to the response. Every + * event that arrives between the capture and the end of the chunked write is + * queued and, under a plain discard, lost outright: nothing re-fetches, and + * the CLI's next partial redraw lands on a frame the terminal never received. + * The caller passes the response's own arrival time as `since` so exactly + * that tail is replayed and the pre-capture events stay dropped. * * COD-144: a brand-new session is the exception. Its terminal fetch can resolve * BEFORE the PTY emits its first prompt, so the fetched buffer is empty and the @@ -3843,14 +3862,22 @@ Object.assign(CodemanApp.prototype, { * After unblocking, new SSE/WS events deliver subsequent output normally. * * @param {string} [owner] Load token from `_beginBufferLoad`; a stale owner is a no-op. - * @param {{ flushQueued?: boolean }} [opts] When `flushQueued` is true, replay any queued events. + * @param {{ flushQueued?: boolean, since?: number }} [opts] When `flushQueued` + * is true, replay queued events whose arrival timestamp is at or after + * `since` (default 0, meaning the whole queue). */ _beginBufferLoad(owner) { if (this._bufferLoadSeq === undefined) this._bufferLoadSeq = 0; const loadOwner = owner === undefined ? `buffer-${++this._bufferLoadSeq}` : owner; + // `selectSession` opens the load before its fetch, and `chunkedTerminalWrite` + // opens it again under the SAME owner when it starts writing. Resetting the + // queue on that second call would throw away everything that arrived during + // the fetch, which on the capture path is output no buffer holds. Re-entering + // one load keeps its queue; a genuinely new load still starts empty. + const reentering = this._bufferLoadOwner === loadOwner && Array.isArray(this._loadBufferQueue); this._bufferLoadOwner = loadOwner; this._isLoadingBuffer = true; - this._loadBufferQueue = []; + if (!reentering) this._loadBufferQueue = []; return loadOwner; }, @@ -3864,9 +3891,13 @@ Object.assign(CodemanApp.prototype, { this._bufferLoadOwner = null; // COD-144: replay (rather than discard) queued live events when the load // painted nothing — the queued prompt is the only content a new session has. + // A tmux-capture load replays too, but only the tail: `since` cuts the queue + // at the moment the capture stopped being able to contain what arrived. if (opts?.flushQueued && queued && queued.length) { - for (const data of queued) { - this.batchTerminalWrite(data); + const since = typeof opts.since === 'number' ? opts.since : 0; + for (const entry of queued) { + if (entry.at < since) continue; + this.batchTerminalWrite(entry.data); } } return true; diff --git a/test/capture-load-window.browser.test.ts b/test/capture-load-window.browser.test.ts new file mode 100644 index 00000000..9f1ee810 --- /dev/null +++ b/test/capture-load-window.browser.test.ts @@ -0,0 +1,177 @@ +/** + * @fileoverview Output arriving after a pane capture survives the buffer load. + * + * `batchTerminalWrite` queues live terminal events while a buffer load runs, + * and `_finishBufferLoad` discards that queue by default. That is right when + * the loaded buffer is the server's accumulated byte history, which is current + * up to the response. A tmux pane capture is current only up to CAPTURE time, + * so anything arriving between the capture and the end of the chunked write is + * queued and then dropped, with nothing scheduling a re-fetch. + * + * The queue now stamps each entry with its arrival time, and a capture load + * replays the tail that arrived after the response headers. These drive the + * real client in chromium: the event is injected from inside the response's + * own `json()` call, which is the one place guaranteed to land after the + * headers and before the chunked write. + * + * Port: 3256 (capture load window) + * + * Run: npx vitest run --config config/vitest.browser.config.ts test/capture-load-window.browser.test.ts + */ + +import { describe, it, expect, beforeAll, afterAll } from 'vitest'; +import { chromium, type Browser, type BrowserContext, type Page } from 'playwright'; +import { WebServer } from '../src/web/server.js'; + +const PORT = 3256; +const BASE_URL = `http://localhost:${PORT}`; +const MARKER = 'ARRIVED-AFTER-THE-CAPTURE'; + +let server: WebServer; +let browser: Browser; + +beforeAll(async () => { + server = new WebServer(PORT, false, true); // testMode + await server.start(); + browser = await chromium.launch({ headless: true }); +}, 60_000); + +afterAll(async () => { + await browser?.close(); + await server?.stop(); +}, 30_000); + +/** + * Select the session with the terminal fetch stubbed, injecting one live event + * from inside `json()`. Returns how many terminal rows carry the marker, so a + * flush that replays too much fails as loudly as one that replays nothing. + */ +async function runLoad(page: Page, sessionId: string, source: string): Promise<number> { + return page.evaluate( + async ({ sid, src, marker }) => { + const app = ( + window as unknown as { + app: { + selectSession: (id: string, o?: object) => Promise<void>; + _onSessionTerminal: (e: { id: string; data: string }) => void; + terminal: { + buffer: { + active: { + length: number; + getLine: (i: number) => { translateToString: (t: boolean) => string } | undefined; + }; + }; + }; + }; + } + ).app; + + const realFetch = window.fetch.bind(window); + window.fetch = ((input: RequestInfo | URL, init?: RequestInit) => { + const url = String(typeof input === 'string' ? input : ((input as Request).url ?? input)); + if (!url.includes('/terminal')) return realFetch(input as RequestInfo, init); + return Promise.resolve({ + ok: true, + status: 200, + // `selectSession` timestamps the headers the moment this promise + // resolves, then calls json(). Injecting here puts the event after + // that timestamp and inside the load window, which is exactly the + // gap a pane capture cannot cover. + json: async () => { + app._onSessionTerminal({ id: sid, data: `\r\n${marker}\r\n` }); + return { + success: true, + data: { + terminalBuffer: '\x1b[1;1Hcaptured frame line one\r\n', + status: 'idle', + fullSize: 512, + retainedBytes: 512, + truncated: false, + truncationReason: null, + source: src, + captureCols: 80, + captureRows: 24, + }, + }; + }, + }) as unknown as Promise<Response>; + }) as typeof window.fetch; + + try { + await app.selectSession(sid); + await new Promise((r) => setTimeout(r, 1200)); + const buf = app.terminal.buffer.active; + let hits = 0; + for (let i = 0; i < buf.length; i++) { + if (buf.getLine(i)?.translateToString(true).includes(marker)) hits += 1; + } + return hits; + } finally { + window.fetch = realFetch; + } + }, + { sid: sessionId, src: source, marker: MARKER } + ); +} + +async function openSession(page: Page): Promise<string> { + await page.goto(BASE_URL, { waitUntil: 'domcontentloaded' }); + await page.waitForFunction(() => document.body.classList.contains('app-loaded'), { timeout: 10_000 }); + // xterm loads from /vendor, so the terminal appears a beat after the app. + // Without it every buffer assertion below would throw rather than compare. + await page.waitForFunction(() => (window as unknown as { app?: { terminal?: unknown } }).app?.terminal, null, { + timeout: 30_000, + }); + return page.evaluate(async () => { + const res = await fetch('/api/sessions', { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ workingDir: '/tmp', name: 'capture-load-window-test' }), + }); + const body = await res.json(); + return body.data?.session?.id ?? body.data?.id ?? body.id; + }); +} + +describe('output emitted during a capture load', () => { + let context: BrowserContext; + let page: Page; + + afterAll(async () => { + await context?.close(); + }); + + it('reaches the terminal exactly once when the buffer came from a pane capture', async () => { + context = await browser.newContext({ viewport: { width: 1280, height: 800 } }); + page = await context.newPage(); + const sessionId = await openSession(page); + expect(sessionId).toBeTruthy(); + + // Exactly once. The cutoff exists so the flush cannot also replay events the + // payload already carried, which would double the output rather than heal it. + expect(await runLoad(page, sessionId, 'mux-visible')).toBe(1); + + await page.evaluate( + (sid: string) => fetch(`/api/sessions/${sid}`, { method: 'DELETE' }).then(() => undefined), + sessionId + ); + await context.close(); + }, 60_000); + + it('stays dropped when the buffer came from the accumulated byte history', async () => { + // The byte history already contains everything up to the response, so + // replaying the queue on top of it would duplicate the output — most + // visibly Ink's cursor-up redraws. The discard has to survive this fix. + context = await browser.newContext({ viewport: { width: 1280, height: 800 } }); + page = await context.newPage(); + const sessionId = await openSession(page); + + expect(await runLoad(page, sessionId, 'history')).toBe(0); + + await page.evaluate( + (sid: string) => fetch(`/api/sessions/${sid}`, { method: 'DELETE' }).then(() => undefined), + sessionId + ); + await context.close(); + }, 60_000); +}); diff --git a/test/terminal-buffer-flush.test.ts b/test/terminal-buffer-flush.test.ts index 54267beb..0500a3e6 100644 --- a/test/terminal-buffer-flush.test.ts +++ b/test/terminal-buffer-flush.test.ts @@ -56,10 +56,10 @@ type BufferLoadApp = { _bufferLoadSeq: number; _bufferLoadOwner: string | null; _isLoadingBuffer: boolean; - _loadBufferQueue: string[] | null; + _loadBufferQueue: { at: number; data: string }[] | null; batchTerminalWrite: (data: string) => void; _beginBufferLoad: (owner?: string) => string; - _finishBufferLoad: (owner?: string, opts?: { flushQueued?: boolean }) => boolean; + _finishBufferLoad: (owner?: string, opts?: { flushQueued?: boolean; since?: number }) => boolean; }; /** @@ -84,10 +84,13 @@ function makeApp() { return { app, writes }; } -/** Simulate live SSE events arriving while a buffer load is in progress (the queue path). */ -function pushWhileLoading(app: BufferLoadApp, data: string) { - // Mirrors batchTerminalWrite's queue branch: if loading, push to the queue. - if (app._isLoadingBuffer && app._loadBufferQueue) app._loadBufferQueue.push(data); +/** + * Simulate a live SSE event arriving while a buffer load is in progress. + * Mirrors batchTerminalWrite's queue branch, which stamps each entry with its + * arrival time so a flush can replay only the tail (see the `since` tests). + */ +function pushWhileLoading(app: BufferLoadApp, data: string, at = performance.now()) { + if (app._isLoadingBuffer && app._loadBufferQueue) app._loadBufferQueue.push({ at, data }); } describe('buffer-load flush (COD-144)', () => { @@ -153,11 +156,91 @@ describe('buffer-load flush (COD-144)', () => { // State untouched — still loading, queue intact, nothing replayed. expect(app._isLoadingBuffer).toBe(true); expect(app._bufferLoadOwner).toBe('real-owner'); - expect(app._loadBufferQueue).toEqual(['queued']); + expect(app._loadBufferQueue).toEqual([{ at: expect.any(Number), data: 'queued' }]); expect(app.batchTerminalWrite).not.toHaveBeenCalled(); expect(writes).toEqual([]); }); + // ── The tmux-capture tail: `since` ── + // + // A pane capture is a point-in-time frame taken part-way through the fetch, so + // it holds what arrived BEFORE the capture and nothing after. selectSession + // passes the response's arrival time as `since`, which splits the queue at + // exactly that line: pre-capture events are already painted and must stay + // dropped, post-capture events exist nowhere else and must be replayed. + + it('flushes only the entries at or after `since`', () => { + const { app, writes } = makeApp(); + const owner = app._beginBufferLoad('load-since'); + pushWhileLoading(app, 'already-in-the-capture', 100); + pushWhileLoading(app, 'arrived-at-the-headers', 200); + pushWhileLoading(app, 'arrived-after-the-headers', 300); + + app._finishBufferLoad(owner, { flushQueued: true, since: 200 }); + + // The pre-capture event stays dropped; the boundary entry counts as after. + expect(writes).toEqual(['arrived-at-the-headers', 'arrived-after-the-headers']); + }); + + it('flushQueued without `since` still replays the whole queue', () => { + // The COD-144 path: a brand-new session's first prompt predates the + // response, so cutting the queue would drop the only content it has. + const { app, writes } = makeApp(); + const owner = app._beginBufferLoad('load-no-since'); + pushWhileLoading(app, 'prompt', 10); + pushWhileLoading(app, 'more', 20); + + app._finishBufferLoad(owner, { flushQueued: true }); + + expect(writes).toEqual(['prompt', 'more']); + }); + + it('a `since` past every entry flushes nothing', () => { + const { app, writes } = makeApp(); + const owner = app._beginBufferLoad('load-since-late'); + pushWhileLoading(app, 'old', 10); + + app._finishBufferLoad(owner, { flushQueued: true, since: 999 }); + + expect(writes).toEqual([]); + expect(app.batchTerminalWrite).not.toHaveBeenCalled(); + }); + + // ── Re-entering one load ── + // + // `selectSession` opens the load before its fetch, and `chunkedTerminalWrite` + // opens it again under the SAME owner when it starts writing. A reset on that + // second call would silently throw away everything queued during the fetch, + // which on the capture path is output no buffer holds. + + it('re-entering the same load keeps what the queue already holds', () => { + const { app, writes } = makeApp(); + const owner = app._beginBufferLoad('load-reenter'); + pushWhileLoading(app, 'arrived-during-the-fetch', 100); + + // chunkedTerminalWrite re-opens the load it was handed. + app._beginBufferLoad(owner); + pushWhileLoading(app, 'arrived-during-the-write', 200); + + app._finishBufferLoad(owner, { flushQueued: true, since: 50 }); + + expect(writes).toEqual(['arrived-during-the-fetch', 'arrived-during-the-write']); + }); + + it('a genuinely different load still starts with an empty queue', () => { + const { app, writes } = makeApp(); + app._beginBufferLoad('load-first'); + pushWhileLoading(app, 'belongs-to-the-abandoned-load', 100); + + // A tab switch starts a new load under a new owner. Its events are not ours. + const second = app._beginBufferLoad('load-second'); + pushWhileLoading(app, 'belongs-to-this-load', 200); + + app._finishBufferLoad(second, { flushQueued: true, since: 0 }); + + expect(writes).toEqual(['belongs-to-this-load']); + }); + it('empty queue + flushQueued is a no-op (no throw, no writes)', () => { const { app, writes } = makeApp(); const owner = app._beginBufferLoad('load-empty'); From 3248f3508105268974e5b60ffee1bb2d9feae1de Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Tue, 15 Sep 2026 18:18:09 +0200 Subject: [PATCH 07/28] chore: version packages (#437) * chore: version packages * chore: sync the CLAUDE.md version line to 1.29.1 Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> --------- Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: Codeman maintainer <noreply@anthropic.com> --- .changeset/auto-name-sessions.md | 11 ----------- .claude-plugin/marketplace.json | 2 +- CHANGELOG.md | 11 +++++++++++ CLAUDE.md | 2 +- package-lock.json | 4 ++-- package.json | 2 +- plugins/codeman/.claude-plugin/plugin.json | 2 +- 7 files changed, 17 insertions(+), 17 deletions(-) delete mode 100644 .changeset/auto-name-sessions.md diff --git a/.changeset/auto-name-sessions.md b/.changeset/auto-name-sessions.md deleted file mode 100644 index 8dbb6f95..00000000 --- a/.changeset/auto-name-sessions.md +++ /dev/null @@ -1,11 +0,0 @@ ---- -"aicodeman": patch ---- - -Auto-name sessions from the first prompt (#376, opt-in). With the new synced **Auto-name Sessions** setting on (App Settings → Appearance → Tabs, default off), a tab that still carries its generated name takes a title from the first real prompt you submit, keeping the case prefix: `w3-myapp` becomes `w3-myapp: fix the login redirect`. The strip shows the title with the prefix in the tooltip, and the next session in that case still counts up. It happens once per session, only for prompts you type or send through the input API (never a Ralph, respawn, cron or approval answer), never for shells, and a name you set yourself is never touched. Slash commands such as `/clear` do not become titles. The title is derived locally from the prompt's first sentence; no text leaves the machine. `nameSource` (`placeholder` / `auto` / `manual`) is a new additive field on session state. - -Landed with the fixes the review of #376 asked for: first prompt only (not every prompt), a user-input gate so Ralph, respawn, cron and approval writes cannot name a tab, the prefix form so the case identity and `w<n>` counter survive, and a keystroke tracker that handles a bare Esc, bracketed pastes, wheel reports, Tab and history recall instead of mis-titling the tab. - -### Thanks - -- @shenlvkang-collab for #376, the auto-naming idea and the ownership plumbing (`nameSource`, the listener wiring, the restore path) it shipped with. diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 9d6c3cb8..1d5433b4 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -10,7 +10,7 @@ "name": "codeman", "source": "./plugins/codeman", "description": "Drive Codeman from inside a Claude Code session: spawn worker sessions, prompt them, wait for them, read their answers, clean up. Acts only inside a Codeman-managed session.", - "version": "1.29.0", + "version": "1.29.1", "author": { "name": "Ark0N", "url": "https://github.com/Ark0N" diff --git a/CHANGELOG.md b/CHANGELOG.md index 3c10b5a5..c2740d9e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,16 @@ # aicodeman +## 1.29.1 + +### Patch Changes + +- 5b920cb: Auto-name sessions from the first prompt (#376, opt-in). With the new synced **Auto-name Sessions** setting on (App Settings → Appearance → Tabs, default off), a tab that still carries its generated name takes a title from the first real prompt you submit, keeping the case prefix: `w3-myapp` becomes `w3-myapp: fix the login redirect`. The strip shows the title with the prefix in the tooltip, and the next session in that case still counts up. It happens once per session, only for prompts you type or send through the input API (never a Ralph, respawn, cron or approval answer), never for shells, and a name you set yourself is never touched. Slash commands such as `/clear` do not become titles. The title is derived locally from the prompt's first sentence; no text leaves the machine. `nameSource` (`placeholder` / `auto` / `manual`) is a new additive field on session state. + + Landed with the fixes the review of #376 asked for: first prompt only (not every prompt), a user-input gate so Ralph, respawn, cron and approval writes cannot name a tab, the prefix form so the case identity and `w<n>` counter survive, and a keystroke tracker that handles a bare Esc, bracketed pastes, wheel reports, Tab and history recall instead of mis-titling the tab. + + ### Thanks + - @shenlvkang-collab for #376, the auto-naming idea and the ownership plumbing (`nameSource`, the listener wiring, the restore path) it shipped with. + ## 1.29.0 ### Minor Changes diff --git a/CLAUDE.md b/CLAUDE.md index 5401e6c8..4ef118e5 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -77,7 +77,7 @@ When user says "COM": CI runs `npm run check:lockfile` on every push/PR, so lockfile drift fails the build even if the `version-packages` script is bypassed. -**Version**: 1.29.0 (must match `package.json`) +**Version**: 1.29.1 (must match `package.json`) ## Project Overview diff --git a/package-lock.json b/package-lock.json index e7293301..22edc788 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "aicodeman", - "version": "1.29.0", + "version": "1.29.1", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "aicodeman", - "version": "1.29.0", + "version": "1.29.1", "hasInstallScript": true, "license": "MIT", "workspaces": [ diff --git a/package.json b/package.json index 791bc3cc..287879ba 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "aicodeman", - "version": "1.29.0", + "version": "1.29.1", "description": "Mission control for AI coding agents - run 20 autonomous agents with real-time monitoring and session persistence", "type": "module", "main": "dist/index.js", diff --git a/plugins/codeman/.claude-plugin/plugin.json b/plugins/codeman/.claude-plugin/plugin.json index 65fd65d6..8d16d73b 100644 --- a/plugins/codeman/.claude-plugin/plugin.json +++ b/plugins/codeman/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "codeman", "description": "Drive Codeman, the self-hosted session manager for AI coding agents, from inside a Claude Code session: spawn worker sessions, prompt them, wait for them, read their answers, clean up. Acts only inside a Codeman-managed session.", - "version": "1.29.0", + "version": "1.29.1", "author": { "name": "Ark0N", "url": "https://github.com/Ark0N" From bd286bf502c9c9dfa8e9be90eeed6ce55e895d1e Mon Sep 17 00:00:00 2001 From: Codeman maintainer <noreply@anthropic.com> Date: Tue, 15 Sep 2026 19:05:59 +0200 Subject: [PATCH 08/28] docs(wiki): catch the manual up to 1.29.0 and add the three run modes it never had The wiki was written for seven run modes and never received Grok Build, DeepSeek Harness or OMP. They now appear everywhere the others do: the modes table and per-CLI notes, install commands, environment prefixes, the Quick Start table, the requirements rows, the vocabulary, and every "seven modes" count. The 1.27 to 1.29.0 changes land on the pages that own them: attaching a case to an existing container, multi-case adoption and the copy-a-case picker (Docker Cases); file reads over ssh in remote cases and what stays unavailable (Remote SSH Sessions, Working With Files, Security); single-page app routing, frame recovery, localhost links as tabs and the egress guard (Web Tabs); DeepSeek as the one non-Claude mode with real stop/blocked signals and Approvals items, Codex's own work detection, last-response, the model-endpoint routes and refreshed counts (HTTP API, Driving From An Agent, Hooks, Notifications, Keeping Agents Running, Core Concepts); Shift+drag, right-click copy, Auto Copy, the Ctrl+Z guard, font weight, the vertical rail and its activity sort (Keyboard Shortcuts, Input And Voice, The Dashboard, Settings Reference); the 600px phone cutoff, Codex shift arrows and iPhone Duo (Mobile Guide); the Docker Compose route and its update rule (Installation, Running As A Service); four new symptom entries and a "which CLIs" question (Troubleshooting, FAQ). Custom model endpoints are deliberately left to #430, which adds that page and edits Agent CLIs, Settings Reference and the sidebar; these edits stay out of the regions #430, #428 and #376 touch, and all three still merge cleanly on top. Both READMEs: the web-tab menu entry is labelled "Add URL" in the UI, not "Add dashboard". Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> --- README.md | 2 +- README.zh-CN.md | 2 +- docs/wiki/Agent-CLIs.md | 96 +++++++++++++++++++--- docs/wiki/Contributing.md | 2 +- docs/wiki/Core-Concepts.md | 20 +++-- docs/wiki/Docker-Cases.md | 33 ++++++-- docs/wiki/Driving-Codeman-From-An-Agent.md | 23 ++++-- docs/wiki/FAQ.md | 6 ++ docs/wiki/HTTP-API.md | 27 ++++-- docs/wiki/Home.md | 8 +- docs/wiki/Hooks-And-Integrations.md | 10 ++- docs/wiki/Input-And-Voice.md | 11 +++ docs/wiki/Installation.md | 30 ++++++- docs/wiki/Keeping-Agents-Running.md | 24 ++++-- docs/wiki/Keyboard-Shortcuts.md | 3 + docs/wiki/Mobile-Guide.md | 12 ++- docs/wiki/Notifications-And-Approvals.md | 10 ++- docs/wiki/Quick-Start.md | 3 + docs/wiki/Remote-SSH-Sessions.md | 15 +++- docs/wiki/Running-As-A-Service.md | 13 +++ docs/wiki/Security.md | 1 + docs/wiki/Settings-Reference.md | 8 +- docs/wiki/The-Dashboard.md | 19 +++-- docs/wiki/Troubleshooting.md | 34 +++++++- docs/wiki/Web-Tabs.md | 23 ++++++ docs/wiki/Working-With-Files.md | 16 ++++ 26 files changed, 371 insertions(+), 80 deletions(-) diff --git a/README.md b/README.md index 31acc908..df93d622 100644 --- a/README.md +++ b/README.md @@ -443,7 +443,7 @@ PTY Output → 16ms Server Batch → DEC 2026 Wrap → SSE → Client rAF → xt - **Clone a GitHub repo as a case** — paste a repository URL into **Add Case → Clone Repo** and Codeman clones it into `~/codeman-cases/<name>` and registers it as a normal case, ready to run an agent in. It preflights the URL while you type (tells you whether it can be cloned anonymously and offers the repo's real branches and tags for the optional branch/tag field), fills the case name in from the URL, and lets you pick which CLI the Run button should use. Public repositories over `https://`; Codeman never collects or stores credentials - **Multi-CLI** — run **Claude Code**, **OpenCode**, **Codex**, **Antigravity**, **Gemini**, **Pi**, **Grok**, **DeepSeek Harness**, or **OMP** per session; env-var prefixes auto-gate (`CLAUDE_CODE_*` vs `OPENCODE_*` vs `CODEX_*` vs `ANTIGRAVITY_*` vs `GEMINI_*`/`GOOGLE_*` vs `PI_*` vs `GROK_*`/`XAI_*` vs `DSH_*`/`DEEPSEEK_*` vs `OMP_*`). See [`docs/opencode-integration.md`](docs/opencode-integration.md), [`docs/pi-integration.md`](docs/pi-integration.md), [`docs/grok-integration.md`](docs/grok-integration.md), [`docs/deepseek-integration.md`](docs/deepseek-integration.md) and [`docs/omp-integration.md`](docs/omp-integration.md) - **Custom model endpoints** _(new in 1.29.0, HTTP API for now)_ — point a session's CLI at any OpenAI-compatible endpoint instead of its native backend: a local llama.cpp, llama-swap, Ollama or vLLM box, or a cloud gateway such as Azure AI Foundry or OpenRouter. Save an endpoint once (`POST /api/model-endpoints`; its models are discovered from `/v1/models`), apply it to a session (`POST /api/sessions/:id/custom-model`), and the CLI restarts in place on that endpoint. Verified live for Claude, OpenCode, Pi, Grok and OMP; Codex, Gemini and DeepSeek have documented gaps, Antigravity has no mechanism. A toolbar picker is the follow-up. See [`docs/custom-model-endpoints.md`](docs/custom-model-endpoints.md) -- **Web tabs** — open Grafana, Uptime Kuma, a Vite dev server or any dashboard URL as a tab beside your sessions (Run dropdown → **Web / URL** → **Add dashboard**). Dashboards are proxied through Codeman's own origin, so an `http://` target works from a phone over HTTPS and through the tunnel, single-page apps route on their own paths, and a frame that reloads recovers itself. A `localhost` link an agent prints opens as a web tab automatically. See [`docs/web-tabs.md`](docs/web-tabs.md) +- **Web tabs** — open Grafana, Uptime Kuma, a Vite dev server or any dashboard URL as a tab beside your sessions (Run dropdown → **Web / URL** → **Add URL**). Dashboards are proxied through Codeman's own origin, so an `http://` target works from a phone over HTTPS and through the tunnel, single-page apps route on their own paths, and a frame that reloads recovers itself. A `localhost` link an agent prints opens as a web tab automatically. See [`docs/web-tabs.md`](docs/web-tabs.md) - **Docker sessions** — run a case inside an isolated, hardened container. One checkbox on **Create New** spins up a container with sensible defaults and starts the agent inside it; multiple sessions share one per-case container, or attach a case to a container you already run; export a container + its workspace to a portable `.tar.gz` to move it to another machine. See [`docs/docker-cases.md`](docs/docker-cases.md) - **Remote SSH sessions** — point a case at another machine and run the agent there inside a durable remote tmux: survives SSH drops, auto-reconnects, and can discover + attach sessions already running on the host; file previews and downloads come over the same ssh connection. See [`docs/remote-sessions.md`](docs/remote-sessions.md) - **Effort & Ultracode** — set a per-session default effort (`low`–`max`) or enable **ultracode** (dynamic multi-agent workflows). Soft defaults only — switchable anytime with `/effort` in-session. Extended-thinking budget is configurable too diff --git a/README.zh-CN.md b/README.zh-CN.md index 8f19c042..ed56cd41 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -445,7 +445,7 @@ PTY 输出 → 16ms 服务端批处理 → DEC 2026 包裹 → SSE → 客户端 - **把 GitHub 仓库克隆成 case** —— 在 **Add Case → Clone Repo** 里粘贴一个仓库 URL,Codeman 会把它克隆到 `~/codeman-cases/<name>` 并注册为普通 case,随时可以跑智能体。输入时它会预检 URL(告诉你能否匿名克隆,并为可选的分支/标签字段提供仓库真实的分支与标签),从 URL 里填好 case 名,还让你选 Run 按钮该用哪个 CLI。支持 `https://` 的公开仓库;Codeman 绝不收集或保存凭据 - **多 CLI** —— 每个会话可选 **Claude Code**、**OpenCode**、**Codex**、**Antigravity**、**Gemini**、**Pi**、**Grok**、**DeepSeek Harness** 或 **OMP**;环境变量前缀自动隔离(`CLAUDE_CODE_*`、`OPENCODE_*`、`CODEX_*`、`ANTIGRAVITY_*`、`GEMINI_*`/`GOOGLE_*`、`PI_*`、`GROK_*`/`XAI_*`、`DSH_*`/`DEEPSEEK_*` 与 `OMP_*`)。详见 [`docs/opencode-integration.md`](docs/opencode-integration.md)、[`docs/pi-integration.md`](docs/pi-integration.md)、[`docs/grok-integration.md`](docs/grok-integration.md)、[`docs/deepseek-integration.md`](docs/deepseek-integration.md) 与 [`docs/omp-integration.md`](docs/omp-integration.md) - **自定义模型端点**(1.29.0 新增,目前仅 HTTP API)—— 让某个会话的 CLI 指向任意 OpenAI 兼容端点,而不是它自己的官方后端:本地的 llama.cpp、llama-swap、Ollama 或 vLLM 机器,也可以是 Azure AI Foundry、OpenRouter 这类云端网关。端点只需保存一次(`POST /api/model-endpoints`,模型列表从它的 `/v1/models` 自动发现),再应用到会话(`POST /api/sessions/:id/custom-model`),CLI 就会在原地重启并接上该端点。Claude、OpenCode、Pi、Grok 与 OMP 已实测通过;Codex、Gemini 与 DeepSeek 存在已记录的缺口,Antigravity 没有可用机制。工具栏选择器是下一步。详见 [`docs/custom-model-endpoints.md`](docs/custom-model-endpoints.md) -- **Web 标签页** —— 把 Grafana、Uptime Kuma、一个 Vite 开发服务器或任何仪表盘 URL 作为标签页打开在会话旁边(Run 下拉菜单 → **Web / URL** → **Add dashboard**)。仪表盘通过 Codeman 自己的源代理,因此 `http://` 目标在手机上走 HTTPS 也能用、走隧道也能用;单页应用能在自己的路径上正常路由,页面自己重载后也能自行恢复。智能体打印出的 `localhost` 链接会自动以 Web 标签页打开。详见 [`docs/web-tabs.md`](docs/web-tabs.md) +- **Web 标签页** —— 把 Grafana、Uptime Kuma、一个 Vite 开发服务器或任何仪表盘 URL 作为标签页打开在会话旁边(Run 下拉菜单 → **Web / URL** → **Add URL**)。仪表盘通过 Codeman 自己的源代理,因此 `http://` 目标在手机上走 HTTPS 也能用、走隧道也能用;单页应用能在自己的路径上正常路由,页面自己重载后也能自行恢复。智能体打印出的 `localhost` 链接会自动以 Web 标签页打开。详见 [`docs/web-tabs.md`](docs/web-tabs.md) - **Docker 会话** —— 在隔离且加固的容器中运行 case。**Create New** 上勾选一个复选框即可用合理的默认值启动容器并在其中启动智能体;同一 case 的多个会话共享一个容器,也可以把 case 挂到你已经在跑的容器上;可将容器连同工作区导出为可移植的 `.tar.gz`,迁移到另一台机器。详见 [`docs/docker-cases.md`](docs/docker-cases.md) - **远程 SSH 会话** —— 把 case 指向另一台机器,让智能体在那里一个持久的远程 tmux 中运行:SSH 断连不中断任务、自动重连,还能发现并附着主机上已在运行的会话;文件预览与下载走同一条 ssh 连接。详见 [`docs/remote-sessions.md`](docs/remote-sessions.md) - **Effort 与 Ultracode** —— 设置每会话的默认 effort(`low`–`max`),或启用 **ultracode**(动态多智能体工作流)。这些都只是软默认值 —— 会话中可随时用 `/effort` 切换。扩展思考预算也可配置 diff --git a/docs/wiki/Agent-CLIs.md b/docs/wiki/Agent-CLIs.md index 6c34f274..eb013b29 100644 --- a/docs/wiki/Agent-CLIs.md +++ b/docs/wiki/Agent-CLIs.md @@ -1,9 +1,9 @@ # Agent CLIs -Codeman drives seven run modes: six agent CLIs plus a plain shell. This page covers picking +Codeman drives ten run modes: nine agent CLIs plus a plain shell. This page covers picking one, setting it up, and the differences that actually change how you work. -## The seven modes +## The ten modes | Mode | CLI | Get it | | -------------------- | ---------------------------- | ---------------------------------------------------------------------- | @@ -13,6 +13,9 @@ one, setting it up, and the differences that actually change how you work. | **Gemini** | `gemini` | [github.com/google-gemini/gemini-cli](https://github.com/google-gemini/gemini-cli) | | **Antigravity** | `agy` | [antigravity.google](https://antigravity.google) | | **Pi** | `pi` | [pi.dev](https://pi.dev) | +| **Grok Build** | `grok` | [github.com/xai-org/grok-build](https://github.com/xai-org/grok-build) | +| **DeepSeek Harness** | `dsh` | [github.com/deepseek-ai/deepseek-harness](https://github.com/deepseek-ai/deepseek-harness) | +| **OMP** | `omp` | [github.com/can1357/oh-my-pi](https://github.com/can1357/oh-my-pi) | | **Terminal / Shell** | your `$SHELL` | Already installed. | Any combination works, including all of them. The run mode is chosen per session from the @@ -47,8 +50,12 @@ If a CLI is installed but a Run button for it never appears: precisely to avoid this; a hand-written plist or unit will not. 3. Restart the server after installing a new CLI. -`pi` is additionally version-probed rather than trusted by name, because `pi` is a generic -enough command that something else on your PATH may answer to it. +`pi`, `grok`, `omp` and `dsh` are additionally identity-probed rather than trusted by name: +`pi` and `omp` are generic enough that something else on your PATH may answer to them, +`grok` has npm squatters, and Debian ships an unrelated `dsh` (dancer's shell). Each has a +status endpoint (`/api/grok/status`, `/api/deepseek/status`, `/api/omp/status`) that reports +the path and version that actually resolved, so a misresolution is visible rather than +presenting as "the mode just does not work". ## Claude is the reference mode @@ -62,15 +69,15 @@ output. The other CLIs expose no equivalent. | Respawn cycling and unattended runs | Yes | Yes | | Cron jobs | Yes | Yes | | Docker cases, remote SSH cases | Yes | Yes | -| Precise idle detection (hook-driven) | Yes | Output-stabilization fallback, coarser | +| Precise idle detection | Yes | Codex: same screen check, via its own prompt and working line. DeepSeek: reports its state itself. Others: output stabilization, coarser | | Auto-resume when a usage limit resets | Yes | No | | Plan usage chip | Yes | No | -| Approvals Inbox | Yes | No | +| Approvals Inbox | Yes | DeepSeek yes; others no | | Read My Mind | Yes | No | | Ralph loop and its task tracker | Yes | No | | Subagent and team windows | Yes | No | | Model, effort, and ultracode controls | Yes | No | -| `stop` and `blocked` wait signals | Yes | 400 if you ask for them explicitly | +| `stop` and `blocked` wait signals | Yes | DeepSeek yes; elsewhere 400 if you ask for them explicitly | | The bundled agent skill | Yes | No | Everything that makes a session a session works everywhere. What is Claude-only is mostly @@ -124,6 +131,11 @@ Two behaviours that are deliberate and worth knowing: - **The wheel is not forwarded** into its transcript. Codex ignores the mouse reports Codeman would send, so forwarding produced a dead wheel. Scrolling in a Codex session is local scrollback. +- **Work detection is Codex's own.** Codex declares its `›` composer glyph and its + `esc to interrupt` working line, so it gets the same screen-checked idle detection Claude + does; before 1.26.1 every Codex session reported idle for its whole life. Codex + conversations also appear in Past Sessions and can be resumed, and on phones the keyboard + bar grows `⇧←` / `⇧→` for Codex's queued-message editing and prompt stack. ### Gemini @@ -157,6 +169,60 @@ Pi needs the opposite instincts from every other CLI here. Guide: [`docs/pi-integration.md`](https://github.com/Ark0N/Codeman/blob/master/docs/pi-integration.md). +### Grok Build + +xAI's `grok`, installed with `curl -fsSL https://x.ai/cli/install.sh | bash` into +`~/.grok/bin`. Codex-shaped on permissions and OpenCode-shaped on rendering: + +- **Its bypass switch is `--always-approve`**, Grok's own `bypassPermissions` mode, and the + Run button sends it the way it sends Codex's. In multi-user mode a user without a grant + has it stripped. +- **Authentication is Grok's own**: browser OAuth on first run (a device-code screen inside + a Codeman pane), `grok login --device-auth` for headless hosts, or `XAI_API_KEY` as a + per-session environment override. +- It renders a full-screen TUI, so scrolling is local scrollback. + +Guide: [`docs/grok-integration.md`](https://github.com/Ark0N/Codeman/blob/master/docs/grok-integration.md). + +### DeepSeek Harness + +The mode wired least like the others, for two reasons worth knowing before you use it. + +**`dsh` is a launcher, not an agent.** It boots a *profile*, and the three DeepSeek ships +(`web`, `headless`, `base`) cannot drive a terminal pane. So "installed" and "runnable" are +different questions: the Run menu offers **DeepSeek** only once a pane-capable profile +exists, and until then shows **DeepSeek — add a terminal profile…**, which installs the +community `dsh-tui` with one click (`pnpm` must be on PATH, because the launcher spawns it +directly). + +**Permissions are an environment variable, not a flag.** The harness has no +skip-permissions switch. `DSH_PERMISSION_MODE` (`read-only`, `workspace-write`, +`danger-full-access`) is the whole control, and it is the one setting Codeman deliberately +carries as an environment variable, because the harness reads it as a soft boot-time +default. In multi-user mode a user without a grant is clamped to `workspace-write`. + +The reward for the odd wiring: **DeepSeek is the one non-Claude mode with real signals.** +Its terminal front door reports idle, working and blocked to Codeman, so a DeepSeek +session gets precise idle detection, the `stop` and `blocked` wait signals, and Approvals +Inbox items. Answers are read from the harness's own transcript on disk rather than +scraped off the pane. The model is not a session setting; it is part of the profile. + +Guide: [`docs/deepseek-integration.md`](https://github.com/Ark0N/Codeman/blob/master/docs/deepseek-integration.md). + +### OMP + +Oh My Pi, installed with `curl -fsSL https://omp.sh/install | sh` into `~/.local/bin`. +OMP owns its auth, provider routing and approval mode entirely in `~/.omp`: there is no +Codeman-side login, key field, or bypass switch. Run `omp` once outside Codeman to finish +its own onboarding, and every session started through Codeman inherits that config. Its +documented default approval mode is `yolo`, so an OMP pane auto-approves tool use with no +flag from Codeman; change that in OMP's own config, not here. + +OMP conversations appear in Past Sessions and can be resumed, and a respawn continues the +same conversation with `--continue`. + +Guide: [`docs/omp-integration.md`](https://github.com/Ark0N/Codeman/blob/master/docs/omp-integration.md). + ### Terminal / Shell A plain shell in a tmux session. No agent, no hooks, no idle detection. @@ -180,9 +246,15 @@ respawns. Which variables are accepted depends on the mode: | Gemini | `GEMINI_*`, `GOOGLE_*` | | Antigravity | `ANTIGRAVITY_*` | | Pi | `PI_*` | +| Grok | `GROK_*`, `XAI_*` | +| DeepSeek | `DSH_*`, `DEEPSEEK_*` | +| OMP | `OMP_*` | Anything outside the allowlist is rejected at the schema. This is intentional: the allowlist -is one global list, so widening it for one CLI widens it for all of them. +is one global list, so widening it for one CLI widens it for all of them. In multi-user mode +the keys that could redirect a CLI's traffic or move its config home (`DSH_PERMISSION_MODE`, +`DSH_HOME`, `DEEPSEEK_BASE_URL`, `OMP_AUTH_BROKER_URL`, and the base URLs and config +directories of the others) are dropped for a user without the bypass grant. Two things that deliberately do **not** travel as environment variables: **effort**, because an environment variable hard-locks it and blocks `/effort`, and **model**, which is written @@ -192,9 +264,11 @@ into the case's `.claude/settings.local.json` so that `/model` keeps working. - **Claude Code** if you want every Codeman feature. Unattended overnight runs, usage-limit auto-resume, the Approvals Inbox, and subagent visualization all assume it. -- **Codex, OpenCode, Gemini, Antigravity** when you prefer that agent or that model. You get - the session layer, respawn, cron, Docker, and remote SSH; you do not get the hook-driven - features. +- **Codex, OpenCode, Gemini, Antigravity, Grok, OMP** when you prefer that agent or that + model. You get the session layer, respawn, cron, Docker, and remote SSH; you do not get the + hook-driven features. +- **DeepSeek Harness** if you want DeepSeek's models with real status signals. It is the one + non-Claude mode that reports idle, working and blocked to Codeman itself. - **Pi** if you want a fast, unsandboxed agent and you understand what project trust does. - **Shell** for the times you want a terminal on your phone with no agent at all. It is a genuinely useful mode, not a fallback. diff --git a/docs/wiki/Contributing.md b/docs/wiki/Contributing.md index f99e6796..892e594f 100644 --- a/docs/wiki/Contributing.md +++ b/docs/wiki/Contributing.md @@ -108,7 +108,7 @@ Conventions for wiki pages: - Images are referenced from the main repository over raw URLs rather than being copied into the wiki. - Say what the default is, especially when it is off. Most of Codeman is opt-in. -- Label Claude-only behaviour every time it appears. Six of the seven run modes are not +- Label Claude-only behaviour every time it appears. Nine of the ten run modes are not Claude. ## Conduct diff --git a/docs/wiki/Core-Concepts.md b/docs/wiki/Core-Concepts.md index 5110e5ef..0afd077d 100644 --- a/docs/wiki/Core-Concepts.md +++ b/docs/wiki/Core-Concepts.md @@ -50,10 +50,10 @@ A session carries state the case does not: ## Run mode The **run mode** is which CLI the session runs: `claude`, `opencode`, `codex`, `gemini`, -`antigravity`, `pi`, or `shell`. It is chosen at start and does not change afterwards; to +`antigravity`, `pi`, `grok`, `deepseek`, `omp`, or `shell`. It is chosen at start and does not change afterwards; to switch, start another session. -Claude is the reference mode. Six of the seven are not Claude, and a number of Codeman +Claude is the reference mode. Nine of the ten are not Claude, and a number of Codeman features are Claude-only for structural reasons rather than missing effort: they depend on Claude Code's hook system or on parsing its terminal output. Every such feature is labelled Claude-only where it appears, and [Agent CLIs](Agent-CLIs) lists them in one place. @@ -68,8 +68,8 @@ Where a case runs is **separate from** which CLI it runs. There are three locati | **Docker** | One long-lived container per case; sessions `docker exec` into it. See [Docker Cases](Docker-Cases). | | **Remote SSH** | A durable tmux server on the remote host, fronted by a local pane running `ssh`. See [Remote SSH Sessions](Remote-SSH-Sessions). | -This matters because it is a common source of confusion: Docker is **not** an eighth run -mode. All seven run modes work in all three locations. A case is docker-backed or +This matters because it is a common source of confusion: Docker is **not** an eleventh run +mode. All ten run modes work in all three locations. A case is docker-backed or ssh-backed; a session is claude or codex or shell. **Web tabs** are the other thing that is not a session. A saved dashboard URL renders as a @@ -155,9 +155,11 @@ report events back: a permission prompt appeared, the turn finished, the agent w task completed. Those events drive tab alerts, the Approvals Inbox, notifications, and the wait primitives. -This is why some features are Claude-only. The other CLIs have no equivalent hook system, -so for them Codeman falls back to watching terminal output, which is coarser: it can see -that something happened, not what it was. +This is why some features are Claude-only. The one partial exception is DeepSeek Harness, +whose terminal front door reports idle, working and blocked to Codeman over the harness's +own supervisor contract, so it gets the hook-driven signals without a hook file. The other +CLIs have no equivalent, so for them Codeman falls back to watching terminal output, which +is coarser: it can see that something happened, not what it was. See [Hooks And Integrations](Hooks-And-Integrations). @@ -167,7 +169,7 @@ See [Hooks And Integrations](Hooks-And-Integrations). | --------------- | ---------------------------------------------------------------------------- | | **Case** | Named working directory. | | **Session** | One CLI in one tmux session. | -| **Run mode** | Which CLI: claude, opencode, codex, gemini, antigravity, pi, shell. | +| **Run mode** | Which CLI: claude, opencode, codex, gemini, antigravity, pi, grok, deepseek, omp, shell. | | **Respawn** | Restarting the CLI on idle to keep an unattended run going. | | **Ralph loop** | An autonomous single-session task loop. | | **Orchestrator**| A phased plan driven across multiple agents. | @@ -178,6 +180,6 @@ See [Hooks And Integrations](Hooks-And-Integrations). ## Read next - [The Dashboard](The-Dashboard) - what the UI is showing you. -- [Agent CLIs](Agent-CLIs) - the seven run modes in detail. +- [Agent CLIs](Agent-CLIs) - the ten run modes in detail. - [Keeping Agents Running](Keeping-Agents-Running) - respawn, idle detection, usage limits. - [`docs/architecture-invariants.md`](https://github.com/Ark0N/Codeman/blob/master/docs/architecture-invariants.md) - the mechanisms behind all of this, for contributors. diff --git a/docs/wiki/Docker-Cases.md b/docs/wiki/Docker-Cases.md index 76c226d1..ff5ccca1 100644 --- a/docs/wiki/Docker-Cases.md +++ b/docs/wiki/Docker-Cases.md @@ -4,7 +4,7 @@ Run a case inside its own container instead of directly on your host: for isolat reproducible toolchain, and for the ability to pick the whole environment up and move it to another machine. -A docker case is a **location overlay**, not a run mode. All seven run modes work inside a +A docker case is a **location overlay**, not a run mode. All ten run modes work inside a container. See [Core Concepts](Core-Concepts). ## One-time setup: the base image @@ -26,7 +26,7 @@ A zero exit code proves the layers ran, not that the toolchain works. Verify: ```bash docker run --rm codeman/agent:base bash -lc \ - 'for c in claude codex gemini opencode agy pi; do printf "%-9s " $c; $c --version 2>&1 | head -1; done' + 'for c in claude codex gemini opencode agy pi grok dsh omp; do printf "%-9s " $c; $c --version 2>&1 | head -1; done' ``` The image is secret-free. Credentials are delivered at runtime, never baked in, so exports @@ -79,6 +79,25 @@ Exactly one long-lived container per case, shared by every session in it. conversation** from the bind-mounted transcript. - Deleting the case removes the container. The workspace on the host survives. +## Attaching to a container you already run + +Tick **Attach to an existing container** on **Add Case → Docker** to link a case to a +container that already exists instead of creating one. Codeman only `exec`s into it and +never creates, starts, stops, restarts or removes it, so a container that is missing or +stopped fails with a message rather than being fixed for you. Drift detection does not +apply (the container carries no Codeman configuration label). The full-image export is +refused, since it would `docker commit` someone else's container, and the workspace export +skips the pause that keeps an owned container consistent during the capture. + +One adopted container can back several cases at different in-container directories, and +**copy an existing case** pre-fills the form from a sibling on the same container. An exact +twin (the same container and the same directory) is refused, as is a container another +user adopted. + +Adoption is **admin-only in multi-user mode**. Linking creates Codeman's own container +with one bind mount that has already been checked; an adopted container's mounts belong to +whoever started it, and one that mounts `/` hands the adopter the host. + ## Credentials Your existing host logins work inside the container without logging in again. Credentials @@ -92,10 +111,12 @@ the container instead. Bind mounts are excluded from image capture, so exports stay secret-free. -One consequence worth knowing: Pi's credentials are seeded per file rather than as a whole -directory, because that directory also holds sessions, extensions, and installed packages, -which can be gigabytes. So in-container Pi sessions are invisible from the host, and `pi -c` -inside a docker case sees only that container's history. +One consequence worth knowing: Pi, Grok and OMP credentials are seeded per file rather than +as whole directories, because those directories also hold sessions, extensions, downloads and +installed packages, which can be gigabytes. So in-container Pi and Grok sessions are +invisible from the host (`pi -c` and `grok -c` inside a docker case see only that +container's history). OMP's `sessions/` is the exception and is shared read-write, because +Codeman reads it host-side for history and resume. ## Isolation diff --git a/docs/wiki/Driving-Codeman-From-An-Agent.md b/docs/wiki/Driving-Codeman-From-An-Agent.md index e324a3af..25faa321 100644 --- a/docs/wiki/Driving-Codeman-From-An-Agent.md +++ b/docs/wiki/Driving-Codeman-From-An-Agent.md @@ -54,7 +54,10 @@ create-time sweep would yank the skill out from under other live sessions sharin directory. Remove them per case with `codeman skill uninstall --case <name>`. The skill ships with the verb index always loaded, plus on-demand references for the verbs, -worked multi-worker recipes, endpoint tables, and cross-session messaging. +worked multi-worker recipes, endpoint tables, and cross-session messaging. It drives +DeepSeek Harness workers the same way it drives Claude ones (`spawn_workers alpha +beta:deepseek` is a mixed fleet in one call), since those are the two modes with real +completion signals. ## The manual path @@ -92,8 +95,9 @@ Read these before writing any code. Each one has cost somebody an afternoon. 5. **Wait instead of polling, and a timeout is not an error.** The wait endpoints answer `200` with `wait.timedOut: true`. Loop over short waits rather than one long call, because tunnels cut idle connections. -6. **Only `claude` sessions emit `stop` and `blocked`.** They come from Claude Code hooks. - Shell and the external CLIs accept only `idle`, `working`, and `exit`; asking for `stop` +6. **Only `claude` and `deepseek` sessions emit `stop` and `blocked`.** Claude's come from + Claude Code hooks, DeepSeek's from the harness reporting its state to Codeman. Shell and + the other external CLIs accept only `idle`, `working`, and `exit`; asking for `stop` explicitly there is a `400`, while omitting `until` is always safe. On a shell session `idle` fires **once at startup and never again**, so synchronize hook-less sessions with an output marker instead. @@ -130,7 +134,10 @@ curl -s -X POST "$API/api/sessions/$ID/input" \ # Or wait for a marker in the output, which works on shell sessions too curl -s "$API/api/sessions/$ID/wait-output?contains=DONE_17909&from=buffer" | jq -# Read the terminal back +# Read the last answer as clean text (claude, codex, deepseek sessions) +curl -s "$API/api/sessions/$ID/last-response" | jq -r '.data.text' + +# Or read the terminal back curl -s "$API/api/sessions/$ID/terminal?tail=4000" | jq -r '.data.output' # Clean up, by exact id @@ -157,7 +164,13 @@ Make it unique per call, because tmux repaints replay old screen text. ### Reading output -Use `terminal?tail=`, not `/output`. The latter's text field is empty for every tmux-backed +For `claude`, `codex` and `deepseek` sessions, read the answer from the transcript rather +than the screen: `GET /api/sessions/:id/last-response` returns the last reply as clean text +with no TUI frames or repaint noise. Poll it briefly rather than reading once, because the +transcript lands slightly after the `stop` signal, so a read immediately after send-and-wait +returns often comes back empty. + +For everything else, use `terminal?tail=`, not `/output`. The latter's text field is empty for every tmux-backed session, which is every interactive session. `tail` counts **bytes**, and what comes back is terminal data with ANSI sequences included. diff --git a/docs/wiki/FAQ.md b/docs/wiki/FAQ.md index bdbdf26b..45c7514b 100644 --- a/docs/wiki/FAQ.md +++ b/docs/wiki/FAQ.md @@ -21,6 +21,12 @@ No. Codeman drives agent CLIs you have already installed and logged in yourself. subscription or key that CLI uses is what pays for the tokens. Codeman never collects, stores, or refreshes your credentials. +### Which agent CLIs does it support? + +Claude Code, OpenCode, Codex, Gemini, Antigravity, Pi, Grok Build, DeepSeek Harness and +OMP, plus a plain shell, chosen per session. Claude is the reference mode and a few features +are Claude-only; [Agent CLIs](Agent-CLIs) has the table. + ### Does Codeman send my code or prompts anywhere? No. There is no telemetry, no analytics, and no phone-home. The only network traffic diff --git a/docs/wiki/HTTP-API.md b/docs/wiki/HTTP-API.md index 46bf7dc6..e31c0dc8 100644 --- a/docs/wiki/HTTP-API.md +++ b/docs/wiki/HTTP-API.md @@ -66,14 +66,14 @@ self-signed certificate, add `-k`. ## Endpoint map -Roughly 200 handlers across 24 route modules. By domain: +Roughly 235 handlers across 26 route modules. By domain: | Domain | Handlers | Covers | | ------------------- | -------- | --------------------------------------------------- | -| System | 45 | Status, settings, search, digest, updates. | -| Sessions | 34 | Create, input, terminal, wait, kill. | -| Cases | 29 | Create, link, clone, remote and docker cases. | -| Files | 16 | Preview, edit, raw, attachments, path picker. | +| System | 56 | Status, settings, digest, updates, tunnel. | +| Sessions | 34 | Create, input, terminal, wait, last response, kill. | +| Cases | 34 | Create, link, clone, remote and docker cases. | +| Files | 17 | Preview, edit, raw, attachments, path picker. | | Orchestrator | 10 | Plans and phases. | | Ralph | 9 | Loop control and configuration. | | Cron | 9 | Jobs and run history. | @@ -82,10 +82,12 @@ Roughly 200 handlers across 24 route modules. By domain: | Respawn | 7 | Respawn configuration and presets. | | Webviews | 6 | Saved dashboards, plus the proxy. | | Mux | 5 | tmux operations. | +| Custom model endpoints | 5 | Saved OpenAI-compatible endpoints, and applying one to a session. | | Push | 4 | Web push subscriptions. | | Read My Mind | 4 | Intent profiles and prediction. | | Scheduled | 4 | The legacy scheduled-run concept. | -| Approvals | 3 | The inbox and answering. | +| Approvals | 4 | The inbox, answering, acknowledging. | +| Tab layout | 2 | Named tab groups per owner. | | Teams, me, search, hooks, clipboard, telemetry, voice, ws | 1-2 each | | Each route module documents its own endpoints in its file header. @@ -114,12 +116,13 @@ Three semantics that break callers who assume otherwise: `wait-output` matches a **literal substring, never a regex.** That is deliberate: no regex means no catastrophic backtracking on attacker-influenced output. -Only `claude` sessions emit `stop` and `blocked`, because those come from Claude Code hooks. -Shell and external CLI sessions accept `idle`, `working`, and `exit`. +Only `claude` and `deepseek` sessions emit `stop` and `blocked`: Claude's come from Claude +Code hooks, DeepSeek's from the harness reporting its state to Codeman. Shell and the other +external CLI sessions accept `idle`, `working`, and `exit`. ## SSE -`GET /api/events` is the live event stream. 156 event names, kept in sync between server and +`GET /api/events` is the live event stream. 158 event names, kept in sync between server and client with a test that fails on drift. The heartbeat is a **named** `sse:heartbeat` event rather than an SSE comment, because @@ -141,6 +144,12 @@ curl -s "$API/api/sessions" | jq '.data[].name' # live sessions curl -s "$API/api/sessions/unified" | jq # live + historical, deduped curl -s "$API/api/subagents" | jq # background agents curl -s "$API/api/search?q=deploy" | jq # cross-session search + +# with ID set to a session id: +curl -s "$API/api/sessions/$ID/last-response" | jq -r '.data.text' # last answer, from the transcript (claude, codex, deepseek) +curl -s "$API/api/model-endpoints" | jq # saved custom OpenAI-compatible endpoints +curl -s -X POST "$API/api/sessions/$ID/custom-model" -H 'Content-Type: application/json' \ + -d '{"endpointId":"local-llama","modelId":"qwen3-27b"}' | jq # restart the CLI on that endpoint; {"clear":true} undoes it ``` ## Limits diff --git a/docs/wiki/Home.md b/docs/wiki/Home.md index 1153d61b..6f469007 100644 --- a/docs/wiki/Home.md +++ b/docs/wiki/Home.md @@ -5,8 +5,8 @@ <h3 align="center">Mission control for AI coding agents</h3> Codeman runs your coding agents on your own machine and puts them behind one dashboard you -can open from any device. It spawns Claude Code, OpenCode, Codex, Antigravity, Gemini, or -Pi inside persistent tmux sessions, streams the real terminal to the browser, and keeps +can open from any device. It spawns Claude Code, OpenCode, Codex, Antigravity, Gemini, Pi, +Grok, DeepSeek Harness, or OMP inside persistent tmux sessions, streams the real terminal to the browser, and keeps working while you are away from the keyboard: it re-prompts idle agents, resumes when a subscription limit resets, runs jobs on a schedule, and shows every background subagent live. @@ -33,7 +33,7 @@ codeman web # then open http://localhost:3000 **Already running it** -- [Agent CLIs](Agent-CLIs) - the seven run modes, their setup, and which features are Claude-only. +- [Agent CLIs](Agent-CLIs) - the ten run modes, their setup, and which features are Claude-only. - [Mobile Guide](Mobile-Guide) - phone and tablet use, QR login, the touch keyboard bar. - [Remote Access](Remote-Access) - Tailscale, Cloudflare tunnel, LAN plus password, QR login. - [Keeping Agents Running](Keeping-Agents-Running) - idle detection, respawn cycling, auto-resume on usage limits. @@ -122,7 +122,7 @@ codeman web # then open http://localhost:3000 | OS | macOS or Linux. Windows works through WSL2. | | Node.js | 22 or newer. | | tmux | Required. Sessions live in tmux, which is what makes them survive restarts. | -| An agent CLI | At least one of Claude Code, OpenCode, Codex, Gemini, Antigravity, Pi. Plain shell sessions need none. | +| An agent CLI | At least one of Claude Code, OpenCode, Codex, Gemini, Antigravity, Pi, Grok Build, DeepSeek Harness, OMP. Plain shell sessions need none. | | Network | Binds to `127.0.0.1` by default. Reaching it from another device is a deliberate step: see [Remote Access](Remote-Access). | Codeman is MIT licensed, self-hosted, and sends no telemetry. Everything runs on your diff --git a/docs/wiki/Hooks-And-Integrations.md b/docs/wiki/Hooks-And-Integrations.md index 08c1efa8..832e0be6 100644 --- a/docs/wiki/Hooks-And-Integrations.md +++ b/docs/wiki/Hooks-And-Integrations.md @@ -19,9 +19,11 @@ terminal into something that can notify you. | `teammate_idle` | An agent-team member goes idle. | Team surfaces. | | `task_completed` | A task finishes. | Task tracking, run summary. | -This is why several Codeman features are Claude-only. The other CLIs have no hook system, so -for them Codeman watches terminal output, which reveals that something happened but not what -it was. +This is why several Codeman features are Claude-only. The one partial exception is DeepSeek +Harness, whose terminal front door reports idle, working and blocked to Codeman over the +harness's own supervisor contract, so it gets the hook-driven surfaces without any hook +file. The other CLIs have no equivalent, so for them Codeman watches terminal output, which +reveals that something happened but not what it was. ### How hooks get installed @@ -67,7 +69,7 @@ sit beside the agents with no code at all. See [Web Tabs](Web-Tabs). ### 2. SSE events `GET /api/events` streams everything Codeman knows: session lifecycle, output, agent -activity, approvals, cron runs. 155 named events, stable under semantic versioning. +activity, approvals, cron runs. 158 named events, stable under semantic versioning. This is the seam for anything that reacts. A bot that pings your chat channel when an agent needs a human is a short script over this stream. diff --git a/docs/wiki/Input-And-Voice.md b/docs/wiki/Input-And-Voice.md index 2942f26a..a873a504 100644 --- a/docs/wiki/Input-And-Voice.md +++ b/docs/wiki/Input-And-Voice.md @@ -27,6 +27,15 @@ The result is the property you want on a phone: a connection that drops mid-prom loses the prompt and never delivers it twice. Two browser tabs on the same session coexist, and only a reconnect from the *same* tab supersedes the old connection. +## Selecting and copying + +Agent CLIs hold the mouse: clicks and drags are reported into the transcript rather than +selecting text. `Shift+drag` starts a selection anyway, right-click copies it (with nothing +selected the native context menu is left alone), and `Ctrl+Shift+C` copies without ever +interrupting. **Auto Copy Selection** in **App Settings → Terminal & Input**, off by +default, copies the moment you release the mouse. On phones, long-press selects; see +[Mobile Guide](Mobile-Guide). + ## Zero-lag local echo On touch devices, keystrokes are painted in the terminal immediately and sent when you press @@ -55,6 +64,8 @@ reconcile against the real buffer and only apply while the cursor is on the comp Chinese, Japanese, and Korean input needs an IME, and an IME needs a real text field. Turning on CJK input in **App Settings → Terminal & Input** puts an always-visible textarea below the terminal that owns composition, then delivers the composed text to the session. +Ctrl- and Alt-modified navigation keys typed through it reach the CLI as the modified +sequences, so word jumps and history keys keep working. ## Voice dictation diff --git a/docs/wiki/Installation.md b/docs/wiki/Installation.md index 476a528c..b8968a4c 100644 --- a/docs/wiki/Installation.md +++ b/docs/wiki/Installation.md @@ -9,7 +9,7 @@ Getting Codeman onto a machine, verifying it works, updating it, and removing it | **macOS or Linux** | Windows works through WSL2. See [Windows](#windows-wsl) below. | | **Node.js 22+** | The installer offers to install it if missing. | | **tmux** | Not optional. Sessions live inside tmux, which is what makes them survive a server restart, a dropped connection, or a closed laptop. | -| **An agent CLI** | At least one of [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), [Gemini CLI](https://github.com/google-gemini/gemini-cli), [Pi](https://pi.dev). Plain shell sessions need none. See [Agent CLIs](Agent-CLIs). | +| **An agent CLI** | At least one of [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), [Gemini CLI](https://github.com/google-gemini/gemini-cli), [Pi](https://pi.dev), [Grok Build](https://github.com/xai-org/grok-build), [DeepSeek Harness](https://github.com/deepseek-ai/deepseek-harness), [OMP](https://github.com/can1357/oh-my-pi). Plain shell sessions need none. See [Agent CLIs](Agent-CLIs). | Codeman itself sends no telemetry and phones no home. The only network traffic is your browser to your server, and whatever the agent CLI you chose does on its own. @@ -20,13 +20,16 @@ browser to your server, and whatever the agent CLI you chose does on its own. curl -fsSL https://getcodeman.com/install | bash ``` -This installs Node.js and tmux if they are missing, clones Codeman into `~/.codeman/app`, -and builds it. +This installs Node.js, tmux and a build toolchain if they are missing (node-pty ships no +Linux prebuild, so it compiles from source), clones Codeman into `~/.codeman/app`, and +builds it. What it asks you: 1. **Permission for every system change.** Package installs and agent CLI downloads are - prompted individually. Nothing is installed silently. + prompted individually. Nothing is installed silently. If no agent CLI is found, a menu + offers to install any of them (DeepSeek excepted: its npm package installs only a + launcher with no runnable profile), or you skip and install one yourself later. 2. **How the dashboard should be reachable.** Three choices: - **Tailscale** (recommended for phone access): keeps the loopback bind and walks you through `tailscale serve`, including the tailnet HTTPS toggle, then verifies the result @@ -101,6 +104,21 @@ at server start, so markup changes need a restart. See [Contributing](Contributing) for the rest of the development loop. +## Route D: Docker Compose + +Codeman itself can run in a container and spawn Docker cases as sibling containers through +the host's Docker socket. Copy `docker/.env.example` to `docker/.env`, set +`CODEMAN_PASSWORD`, then: + +```bash +bash docker/Start-Codeman.sh +``` + +Run the script again after updating rather than a plain `docker compose up`, so the rebuilt +image, the refreshed volumes and the entrypoint arrive together. The full guide, including +storage and networking options, is +[`docker/README.md`](https://github.com/Ark0N/Codeman/blob/master/docker/README.md). + ## Installing an agent CLI Codeman drives CLIs, it does not bundle them. Install at least one: @@ -113,6 +131,9 @@ Codeman drives CLIs, it does not bundle them. Install at least one: | **Antigravity** | See [antigravity.google](https://antigravity.google) | Google's successor to the consumer Gemini CLI. | | **Gemini CLI** | See [github.com/google-gemini/gemini-cli](https://github.com/google-gemini/gemini-cli) | Enterprise only since Google's June 2026 consumer cutover. | | **Pi** | See [pi.dev](https://pi.dev) | No permission prompts and no sandbox by design. Read [Agent CLIs](Agent-CLIs) before using it on a repo you care about. | +| **Grok Build** | `curl -fsSL https://x.ai/cli/install.sh \| bash` | xAI. Lands in `~/.grok/bin`; `grok login --device-auth` for headless hosts. | +| **DeepSeek Harness** | `npm i -g @deepseek-ai/dsh pnpm`, then a terminal profile | The npm package is only a launcher. Codeman's Run menu installs the community terminal profile for you. See [Agent CLIs](Agent-CLIs). | +| **OMP** | `curl -fsSL https://omp.sh/install \| sh` | Oh My Pi. Run it once by hand to finish its own onboarding. | Log each CLI in once, by hand, before pointing Codeman at it. Codeman never collects or stores your CLI credentials. @@ -161,6 +182,7 @@ Full detail, including logs and the self-updater, is in | Installer | Re-run the one-liner, or **App Settings → System → Updates** in the UI. | | npm | `npm update -g aicodeman` | | git clone | `git pull && npm install && npm run build`, then restart. | +| Docker Compose | Re-run `Start-Codeman.sh`. The in-app updater works too, and refuses a release that changes the container definition until you re-run the script. | The in-app updater covers git-clone installs supervised by systemd or launchd. It restarts the process that is running it, so the actual work happens in a detached script and the diff --git a/docs/wiki/Keeping-Agents-Running.md b/docs/wiki/Keeping-Agents-Running.md index 8ba04e8a..68886a5a 100644 --- a/docs/wiki/Keeping-Agents-Running.md +++ b/docs/wiki/Keeping-Agents-Running.md @@ -27,9 +27,12 @@ keystroke echo. Idle now lands a few seconds after a turn genuinely ends. There are several layers stacked on that: a completion message from the CLI, an AI check, output silence, and token stability. -**For every other CLI**, there are no hooks to lean on, so detection is output -stabilization: the session is idle when output stops changing. Coarser, and it is why the -features further down this page are Claude-only. +**For the other CLIs** it depends on what the CLI tells Codeman. Codex declares its own +prompt glyph and working line, so it gets the same screen check Claude does (before 1.26.1 +every Codex session reported idle for its whole life). DeepSeek Harness reports idle, +working and blocked to Codeman itself, which is as precise as hooks. Everything else is +output stabilization: the session is idle when output stops changing. Coarser, and it is +why the features further down this page are Claude-only. ## The Respawn Controller @@ -101,13 +104,18 @@ subscription plan. **Claude only.** A header chip showing live subscription usage, on by default on desktop and off on phones. -It works by installing a status line exporter into Claude Code, which posts Claude's own -rate limit data back to Codeman. The exporter is marker-identified, so it only ever touches -a status line Codeman installed, never one you wrote yourself, and it prints your footer -through so the in-terminal status line still works. +It works through a status line exporter that Codeman hands to `claude` as an ephemeral +setting when it spawns the session, never written to disk, which posts Claude's own rate +limit data back to Codeman. Your own status line (project-local, project, then +`~/.claude/settings.json`) is wrapped and printed through, and a `claude` you run by hand +outside Codeman sees nothing of it. Workspaces an older Codeman wrote the exporter into are +cleaned up the first time a session starts there. Codex limits come from a read-only poll of +its own app-server. Known limit: sessions inside a Docker case do not feed the chip yet. The chip and the exporter are the same setting. Turning the chip on without the exporter -would leave it showing a dash forever, so resolve it in one place: **App Settings**. +would leave it showing a dash forever, so resolve it in one place: **App Settings**. A +device writes the switch only when it flips the chip, so a phone (chip off by default) +saving its font size cannot switch collection off for your desktop. ## Circuit breakers diff --git a/docs/wiki/Keyboard-Shortcuts.md b/docs/wiki/Keyboard-Shortcuts.md index e0c15b3c..bfa32d5d 100644 --- a/docs/wiki/Keyboard-Shortcuts.md +++ b/docs/wiki/Keyboard-Shortcuts.md @@ -30,6 +30,9 @@ Press `Ctrl+?` in the app for the same list in a floating overlay. | `Ctrl+Shift+R` | Restore terminal size. | | `Ctrl` `+` / `Ctrl` `-` | Font size. | | `Shift+Wheel` | Scroll the local buffer, even where the wheel is forwarded to the CLI. | +| `Shift+drag` | Start a selection in a pane whose mouse events go to the CLI. | +| Right-click | Copy the selection. With nothing selected the native menu is left alone. | +| `Ctrl+Z` | Swallowed in agent sessions so a running CLI cannot be suspended. Normal job control in a shell. | ## Everything else diff --git a/docs/wiki/Mobile-Guide.md b/docs/wiki/Mobile-Guide.md index 9489c55e..1751db5c 100644 --- a/docs/wiki/Mobile-Guide.md +++ b/docs/wiki/Mobile-Guide.md @@ -30,8 +30,12 @@ require a secure context. | Toolbar | Bottom: Run, Stop, **Enter**, case picker, voice, settings. | | Keyboard bar | Above the on-screen keyboard when it is open. | -Layout respects notch and home-indicator safe areas, touch targets are 44px, and the case -picker is a bottom sheet rather than a dropdown. +The phone layout applies up to 599px of viewport width, so the Plus and Pro Max iPhones, +the Pixel Pro and a folded Z Fold get it too; wider devices get the tablet layout. Layout +respects notch and home-indicator safe areas, touch targets are 44px, and the case picker is +a bottom sheet rather than a dropdown. On a folding phone (iPhone Duo) dialogs stay clear of +the hinge, and opening or closing the device is treated as the device changing shape, never +as the keyboard appearing. **Swipe left and right** on the terminal to switch sessions. @@ -58,7 +62,9 @@ A row of keys above the virtual keyboard, and what it contains depends on the se **Agent sessions** get quick actions: `/init`, `/clear`, `/compact`, a clipboard key, `Esc`, a path picker, an image key, and 🧠 when Read My Mind is on. Destructive commands need a -double press, so you cannot fire `/clear` with a stray thumb. +double press, so you cannot fire `/clear` with a stray thumb. On Codex sessions the bar also +shows `⇧←` and `⇧→`, the Shift-modified arrows Codex binds to editing the last queued +message and walking the prompt stack. **Shell sessions** automatically swap it for terminal controls: `Ctrl`, `Esc`, `Tab`, four arrows, paste, and dismiss. Your normal preference is remembered and restored when you diff --git a/docs/wiki/Notifications-And-Approvals.md b/docs/wiki/Notifications-And-Approvals.md index ed7de7f5..c096cd50 100644 --- a/docs/wiki/Notifications-And-Approvals.md +++ b/docs/wiki/Notifications-And-Approvals.md @@ -33,8 +33,8 @@ reloading the dashboard while a permission dialog is blocking a session does not with a normal-looking tab. For Claude sessions, these come from Claude Code's hooks and are precise about *why* the -session stopped. For other CLIs there are no hooks, so you get the coarser output-based -signal. +session stopped; DeepSeek Harness sessions report the same states themselves. For the other +CLIs there are no hooks, so you get the coarser output-based signal. ## Window title and OS notifications @@ -62,7 +62,8 @@ Once subscribed, a blocking prompt reaches your phone even from a locked screen. ## The Approvals Inbox -**Opt-in, off by default. Claude sessions only.** +**Opt-in, off by default. Claude sessions, plus DeepSeek Harness sessions, whose terminal +front door reports its prompts to Codeman.** One queue of every prompt currently waiting on a human, across all your sessions, answerable in place. When you have eight workers running, this is the difference between checking eight @@ -136,7 +137,8 @@ from the lock screen. - **No push over plain HTTP.** It is a browser requirement, not a Codeman one. - **iOS needs the home screen install.** A Safari tab will never receive push. - **The bell is invisible at zero.** That is deliberate, not a broken setting. -- **Approvals are Claude-only.** They are built on hook events the other CLIs do not emit. +- **Approvals need real signals.** They are built on hook events, which Claude emits and + DeepSeek Harness reports itself; the other CLIs do neither. - **A stale menu answer is refused, not sent.** If you answer a card for a dialog that has since gone away, Codeman declines rather than typing a digit into the composer. diff --git a/docs/wiki/Quick-Start.md b/docs/wiki/Quick-Start.md index 91e1fc04..2256b545 100644 --- a/docs/wiki/Quick-Start.md +++ b/docs/wiki/Quick-Start.md @@ -67,6 +67,9 @@ one: | **Gemini** | Enterprise only since Google's consumer cutover. | | **Antigravity** | Google's successor to the consumer Gemini CLI. | | **Pi** | No permission prompts and no sandbox by design. | +| **Grok Build** | xAI's CLI. | +| **DeepSeek Harness** | Needs a terminal profile; the menu offers to install one. | +| **OMP** | Oh My Pi, configured entirely through its own `~/.omp`. | | **Terminal / Shell** | A plain shell, no agent. Also the **Run Shell** button. | The dropdown also lists any saved dashboard URLs ([Web Tabs](Web-Tabs)) and your recent diff --git a/docs/wiki/Remote-SSH-Sessions.md b/docs/wiki/Remote-SSH-Sessions.md index 77b8f67a..d6e63b61 100644 --- a/docs/wiki/Remote-SSH-Sessions.md +++ b/docs/wiki/Remote-SSH-Sessions.md @@ -4,7 +4,7 @@ Point a case at another machine and the agent runs **there**, with the same dash mobile UI, and autonomy features. Your laptop becomes a window onto a session living on the remote host. -Like Docker, this is a **location overlay** on a case, not a run mode. All seven run modes +Like Docker, this is a **location overlay** on a case, not a run mode. All ten run modes work remotely. See [Core Concepts](Core-Concepts). ## Why bother @@ -52,7 +52,10 @@ A watcher with bounded backoff notices a dead SSH pane and quietly reattaches to running remote session. On by default; the kill switch is in **App Settings → Agents & CLIs → Remote auto-reconnect**. -Intentional kills are never revived. Closing a session means closing it. +Intentional kills are never revived. Closing a session means closing it. Neither is a clean +exit inside the pane (Ctrl-D, `exit`, Ctrl-C at the CLI's prompt): that tears the remote +tmux session down, and the watcher revives a session only when that durable session is +verifiably still alive. Only a transport drop is reconnected. ## Discover and attach @@ -70,6 +73,14 @@ Attaching to someone else's session and closing your tab must not end their run, not. Several clients can attach the same remote session at different window sizes without clamping each other, and discovery shows a shared badge with the client count. +## Files + +Previews, downloads and text reads in a remote case go over the same ssh connection the +session uses, so a clicked path opens the file on the machine the agent is on, `Range` +seeking included. Nothing is copied to the Codeman host. Editing, Office previews, +thumbnails, the file tree and the tail viewer are not available remotely and answer a clear +400 rather than a misleading 404. Details in [Working With Files](Working-With-Files). + ## Security Every SSH command line in Codeman flows through one builder that shell-escapes every diff --git a/docs/wiki/Running-As-A-Service.md b/docs/wiki/Running-As-A-Service.md index 17d7c49f..616ac0ea 100644 --- a/docs/wiki/Running-As-A-Service.md +++ b/docs/wiki/Running-As-A-Service.md @@ -139,6 +139,7 @@ log stream --predicate 'process == "node"' # macOS, noisy | Installer | Re-run the one-liner, or **App Settings → System → Updates**. | | npm | `npm update -g aicodeman` | | git clone | `git pull && npm install && npm run build`, then restart. | +| Docker Compose | Re-run `Start-Codeman.sh`, or the in-app updater, which restarts the container in place. | ### The in-app updater @@ -170,6 +171,18 @@ service without colliding with the main one. `CODEMAN_DATA_DIR` and `CODEMAN_TMU exist for the rare case where they need to differ, but setting only one of them recreates exactly the problem you were avoiding. +## Running Codeman itself in Docker + +The Compose deployment in `docker/` runs the server in a container and spawns Docker cases +as sibling containers through the mounted host socket. Start it with +`bash docker/Start-Codeman.sh` rather than a bare `docker compose up`: the script pre-creates +the bind-mounted directories with the right owner, honours a `docker-compose.override.yml`, +and refreshes the build volumes when the checkout moved under them. The in-app updater +applies code only and restarts by letting the container exit, so it refuses a release that +changes the Dockerfile, the compose file, or adds a new `.env` key, until you re-run the +script. Guide: +[`docker/README.md`](https://github.com/Ark0N/Codeman/blob/master/docker/README.md). + ## The tunnel as a service ```bash diff --git a/docs/wiki/Security.md b/docs/wiki/Security.md index 2e16511e..8dbaaa38 100644 --- a/docs/wiki/Security.md +++ b/docs/wiki/Security.md @@ -67,6 +67,7 @@ be wrong for at least one of them: | **File Viewer** | Real path resolution before boundary checks, so symlinks cannot escape. Sensitive trees blocked. Edit mode adds an extension allowlist, a size cap, `.git` denial, and optimistic concurrency. It never creates files. | | **Attachments** | An id-based registry, so browser requests never carry absolute paths. The magic-link scanner is prompt-injectable by nature and is therefore force-confined to the session's workspace. Extension allowlist, not a blocklist. | | **Path picker** | Its own root allowlist rather than the workspace confinement. In multi-user mode a non-admin gets only their own user space, because per-user spaces live inside the home directory. | +| **Remote cases** | Reads go over the session's own ssh connection and are resolved and contained on the remote host, with a bounded number of ssh children. Nothing is copied to the Codeman host; writes, Office previews and thumbnails are refused. | Downloads block sensitive paths outright (`.env`, credentials files, `~/.ssh`, AWS credentials), and SVG and HTML are served as downloads with `nosniff` so they cannot execute diff --git a/docs/wiki/Settings-Reference.md b/docs/wiki/Settings-Reference.md index a23a44c6..3255f56f 100644 --- a/docs/wiki/Settings-Reference.md +++ b/docs/wiki/Settings-Reference.md @@ -46,6 +46,7 @@ supervised by systemd or launchd; npm installs report as non-updatable. See | Extended Keyboard Bar | Per device | Which accessory bar phones get. Shell sessions override it while they are active. | | Wheel Scrolls Local History | Off | Keeps the wheel on the local buffer instead of forwarding it to the CLI. | | Auto Copy Selection | Off | Copies highlighted terminal text to the clipboard the moment you finish selecting it. Ctrl+C still copies on demand. | +| Normal / Bold font weight | xterm defaults | Per device, each slot from 100 to 900. The bundled JetBrains Mono renders every step, so a lighter normal weight makes Claude's bold headings stand out. Applies live to the terminal, both echo overlays and open team panes. | | WebGL Renderer | On | With a GPU-stall watchdog that falls back to DOM rendering. | | Gesture Control | Off | Camera hand tracking. Also needs `CODEMAN_GESTURE=1` on the server. | @@ -72,7 +73,9 @@ every session or only the active tab. | Entrance Animations | Per-surface animation styles for tabs, terminals, windows, and lineage lines. All default to the legacy no-animation behaviour. | | Display Name | Your name in the UI. Cosmetic only; it never renames the package, CLI, API, or storage. | | Interface Language | English or Simplified Chinese. Per device. | -| Session List Layout | Header tab strip (default) or a collapsible left sidebar. See [The Dashboard](The-Dashboard#session-list-layout). | +| Session List Layout | Header tab strip (default), a collapsible left sidebar, or the sidebar with detailed rows. See [The Dashboard](The-Dashboard#session-list-layout). | +| Tab Orientation | Keeps the header list but turns the strip vertical beside the terminal, resizable, with detailed rows by default. Desktop and tablet only. | +| Vertical Rail Order | *By activity* (default) sorts the rail the way the home screens are sorted; *Manual* keeps your tab order and drag-reordering. | | Tall Tabs | Taller tab strip. | | Pop-out Button on Tabs | Adds the detach control to tabs, with a per-tab override. | | Spawn Lineage Lines | Arcs from a parent tab to sessions it spawned. Desktop only, on by default. | @@ -156,6 +159,9 @@ Some things are configured before the server starts, not in the UI: | `CODEMAN_DOCKER_BRIDGE_HOOKS` | Lets in-container hooks reach the host on a loopback bind. | | `CODEMAN_FILE_PICKER_ROOTS` | Extra roots for the path picker. | | `CODEMAN_ALLOW_UNAUTHENTICATED_NETWORK` | Acknowledges exposing the server with no password. | +| `CODEMAN_BASE_URL` | Mounts Codeman under a sub-path behind a reverse proxy that forwards the prefix unchanged. See [Remote Access](Remote-Access). | +| `CODEMAN_MAX_DOWNLOAD_BYTES` | Cap on raw file bodies and downloads. 2 GB by default, `0` for none. | +| `CODEMAN_MAX_REMOTE_FILE_SSH` | Concurrent ssh reads for files in remote cases. 4 by default. | ## Gotchas diff --git a/docs/wiki/The-Dashboard.md b/docs/wiki/The-Dashboard.md index 5672b70a..a0e86e5c 100644 --- a/docs/wiki/The-Dashboard.md +++ b/docs/wiki/The-Dashboard.md @@ -22,12 +22,14 @@ page says so and names the setting. The session list lives in the header as a horizontal strip by default. With a lot of sessions open that strip stops being scannable, so **App Settings → Appearance → Tabs → -Session List Layout** can move it into a vertical sidebar on the left instead. +Session List Layout** can move it into a vertical sidebar on the left instead, and +**Tab Orientation** can turn the strip itself into a vertical rail. | Layout | Behaviour | | -------------------- | --------------------------------------------------------------------------------- | | **Header tab strip** | The default. Wraps to a second row on desktop, scrolls sideways on a phone. | -| **Left sidebar** | A vertical list with a filter box and a live session count. `Alt+B` collapses it to a narrow rail that keeps the status dots and task badges visible. On a phone it is an off-canvas drawer rather than a docked rail. | +| **Left sidebar** | A vertical list with a filter box and a live session count. `Alt+B` collapses it to a narrow rail that keeps the status dots and task badges visible. On a phone it is an off-canvas drawer rather than a docked rail. A detailed variant adds the home screen's per-session line (`created 3d ago · working 12m`) and a status pill. | +| **Vertical rail** | The strip turned vertical beside the terminal, resizable, with detailed rows by default. **Vertical Rail Order** sorts it by activity (blocked on you first, then longest running, then most recently quiet), the same order as the home screens; pick *Manual* to get your own order and drag-reordering back. Desktop and tablet only. | It is the same list either way, just re-hosted: tab order, drag-to-reorder, the `Alt+1` to `Alt+9` numbers and every status colour below behave identically in both. The setting is @@ -154,6 +156,10 @@ Worth knowing: always local scrollback. Other CLIs scroll locally. - **Selection copy.** `Ctrl+C` copies when text is selected and interrupts when it is not. `Ctrl+Shift+C` always copies. +- **Selecting where the CLI owns the mouse.** `Shift+drag` starts a selection even in a pane + whose mouse events are forwarded to the CLI, and right-click copies the selection (with + nothing selected the native menu is left alone). **Auto Copy Selection** in App Settings + copies the moment you release. - **Zero-lag input.** On touch devices, keystrokes paint locally before the round trip. See [Input And Voice](Input-And-Voice). - **Renderer.** WebGL by default, with a watchdog that falls back to DOM rendering if the @@ -167,8 +173,9 @@ which lists past sessions including Claude conversations started outside Codeman Two extras depending on the device: -- **Desktop, wide windows**: your open tabs appear as a rail docked to the left edge, in tab - order, with created and last-active stamps. It needs at least 1180px of width; below that +- **Desktop, wide windows**: your open tabs appear as a rail docked to the left edge, in + overview order (blocked on you first, then longest running, then most recently quiet), + with created and state-duration stamps. It needs at least 1180px of width; below that it is hidden so it cannot overlap the search panel. - **Phones**: tapping the "C" logo gives a session overview instead: NEEDS YOU first, then current sessions, then past ones. On by default. @@ -204,7 +211,9 @@ so it is fast and cannot be turned into a traversal. ## Appearance **App Settings → Appearance** carries the theme skins, including light ones. The choice is -applied before the first paint, so there is no flash of the wrong theme on load. +applied before the first paint, so there is no flash of the wrong theme on load. Terminal +font family and weight are per device too: a normal and a bold weight, each from 100 to +900, and the bundled JetBrains Mono renders every step. The same section has the entrance animations for tabs, terminals, agent windows, and lineage lines. All of them default to the legacy no-animation behaviour, so an untouched diff --git a/docs/wiki/Troubleshooting.md b/docs/wiki/Troubleshooting.md index 75b3c768..7eb279b3 100644 --- a/docs/wiki/Troubleshooting.md +++ b/docs/wiki/Troubleshooting.md @@ -138,6 +138,12 @@ That is the PTY-exit circuit breaker. Repeated rapid PTY exits trip it, and it b automatic restarts so a broken configuration does not spin forever. Reset it explicitly from the session's controls. Reattaching does not clear it, deliberately. +### Typed prompts are silently ignored after restoring a tab + +Update. A browser whose input sequence counter fell behind the server's (a restored tab, +cleared site data) used to have every prompt deduplicated away. Since 1.29.0 the duplicate +acknowledgement carries the watermark and the client re-sends. + ### Sessions I did not create appeared, or my session resized itself Two Codeman servers are running against the same data directory and tmux socket. The second @@ -167,6 +173,16 @@ Things to try: Codex ignores the mouse reports that forwarding would send, so Codeman does not forward there. Scrolling is local, and `Shift+Wheel` behaves the same way. +### Selected text is invisible on a light skin + +Update. Every skin named its selection colour under a key xterm renamed in v5, so the four +light skins painted white at 30% over near-white. Fixed in 1.29.0. + +### `Ctrl+Z` suspended my agent + +Update. Since 1.28.0 `Ctrl+Z` is swallowed in agent sessions, so a running CLI cannot be +stopped by job control. Shell sessions keep it. + ### `Ctrl+C` copies when I wanted to interrupt With a selection, `Ctrl+C` copies. With no selection, it interrupts. Clear the selection @@ -252,11 +268,25 @@ node scripts/build-agent-image.mjs --no-cache A plain rebuild reuses the cached `npm install -g` layer and keeps the CLIs frozen at their original versions while reporting success. +### Every file in a remote case says "File not found" + +Update. Before 1.29.0 the file routes resolved every path on the Codeman host, so in a +remote case every click failed while the file plainly existed on the other machine. Reads +now go over ssh; see [Working With Files](Working-With-Files). Editing and Office previews +stay unavailable remotely and say so with a 400. + +### Compose: the server crash-loops with `EACCES` on first start + +Start the stack with `bash docker/Start-Codeman.sh` rather than a plain `docker compose up`, +and update: since 1.29.0 the entrypoint corrects a root-owned bind mount before dropping +privileges. See [Running As A Service](Running-As-A-Service). + ### A remote SSH session dropped and did not come back A bounded-backoff watcher reattaches dropped sessions, and it is on by default. Intentional -kills are never revived. Check the host is reachable and that the remote tmux server is -still running. +kills are never revived, and neither is a clean exit inside the pane (Ctrl-D, `exit`): only +a transport drop is reconnected. Check the host is reachable and that the remote tmux server +is still running. ## Gathering diagnostics diff --git a/docs/wiki/Web-Tabs.md b/docs/wiki/Web-Tabs.md index 76744dad..b22e5010 100644 --- a/docs/wiki/Web-Tabs.md +++ b/docs/wiki/Web-Tabs.md @@ -24,6 +24,23 @@ Switching tabs does not reload a dashboard. Frames stay alive in the background, took a while to authenticate is still there when you come back. Past six live frames, the least recently viewed is dropped to bound memory. +## Single-page apps, reloads and links + +A history-routed dashboard (React Router, Vue Router, a Vite dev server) sees the path it +would see on its own origin, not the proxy prefix, so it renders its real route instead of +its own "page not found". A navigation the page starts itself afterwards, a dev server's +full reload or a root-absolute `location.href`, would land outside the proxy with no +capability; Codeman recognises it, answers with a small recovery page, and remounts the +frame at the path that was lost, bounded to five recoveries a minute per frame. A reload on +the dashboard's landing page is recovered the same way. + +A `localhost` or `127.0.0.1` link in agent output opens as a web tab automatically, reusing +a saved dashboard for the same server or saving one under its `host:port`. On a phone that +address only exists on the Codeman box, so the link would otherwise be a guaranteed +connection error. LAN and tailnet addresses still open directly. `*.localhost` names are +deliberately not auto-routed: they are DNS names rather than address literals, and the link +came from agent output. Add such a dashboard by hand instead. + ## Why dashboards are proxied A plain cross-origin iframe fails three ways at once in the setup Codeman actually ships in: @@ -89,6 +106,12 @@ The proxy authenticates on an in-memory capability embedded in the path, which i exempt from the cookie and Origin checks that every API route enforces. That exemption is fenced to safe methods and non-API paths, and there is a test pinning it in place. +Saved URLs are refused when they point at a link-local or cloud-metadata address, at save +time and again against the address the name resolves to at connect time; loopback and +private ranges stay allowed, because a `localhost` Grafana is the feature. Capabilities are +revoked on logout, and proxied responses carry a same-origin referrer policy so a dashboard +cannot hand the capability-bearing URL to a third party. + Two failure modes that only appear inside a sandboxed frame, and that curl can never reproduce, are handled: runtime-built root-absolute URLs escaping the injected base, and same-host requests being CORS-checked with a null origin. Both present as the dashboard's own diff --git a/docs/wiki/Working-With-Files.md b/docs/wiki/Working-With-Files.md index b579c4cd..d9ebaf48 100644 --- a/docs/wiki/Working-With-Files.md +++ b/docs/wiki/Working-With-Files.md @@ -110,6 +110,22 @@ it is written. Outside the workspace they open in the preview instead: the tail Nothing is registered until you click. Opening a file this way does not add an attachment card. +## Remote (SSH) cases + +In a remote case the workspace lives on the other machine, and so do the files. Previews, +downloads, text reads and the clicked-path route all go over the same ssh connection the +session uses: one `realpath` plus `stat` probe for the file and the workspace root, then a +streamed `cat` (or a slice of it, so video seeking works). Symlinks are resolved on the host +that can resolve them, the size cap applies to the remote size before a byte is requested, +and an unreachable host answers 502 rather than pretending the file is missing. Nothing is +ever copied onto the Codeman host, and a same-named local file is never served under a +remote name. + +Not available over ssh, and said so with a 400 instead of a misleading 404: editing in +place, Office previews and generated thumbnails (both need the bytes on the server's disk), +the file tree and path picker, and the tail viewer. Docker cases are unaffected, because +their workspace is bind-mounted at the same path. + ## The path picker For choosing a path rather than typing one. It appears in two places: From a1c35da0d8edb8f948905bfb07a17f92326cc98b Mon Sep 17 00:00:00 2001 From: codeman-local <codeman@local> Date: Wed, 16 Sep 2026 09:26:30 +0800 Subject: [PATCH 09/28] fix(input): deliver a recovered keystroke before the Enter that submits it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every message typed on an Android phone lost its last character. An Android soft keyboard commits the last typed character and sends the Enter key in ONE InputConnection transaction, so the committed-text `input` event and the Enter keydown are both processed before any zero-delay timer runs. The orphaned-input recovery from #388 resolved its candidate only on such a timer, and that lost the character twice over: * ORDER — xterm emits `\r` synchronously from the Enter keydown, and the local-echo composer submits `pendingText` right there. The recovered character arrived one macrotask too late to be part of the prompt. * LOSS — that same `\r` bumps the canonical counter, so by the time the candidate resolved, `canonicalCount > snapshot` read as "xterm spoke for this keystroke" and stood the recovery down. The character was not merely late, it was dropped. Drain pending candidates synchronously at the next keydown instead, from xterm's custom key handler, which runs before xterm processes that key. The counter then still holds the value it had while the candidate's own keystroke was current, so the stand-down decision is made against the right keystroke, and the recovered byte reaches the composer ahead of whatever the new key emits. The timer stays as the fallback for a keystroke with no key after it. Physical keyboards are unaffected: there the timer has already resolved the candidate long before the next key arrives. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- .changeset/android-last-character-on-enter.md | 5 ++ .../public/terminal-keycode229-recovery.js | 37 +++++++++++++++ test/terminal-keycode229-recovery.test.ts | 46 +++++++++++++++++++ 3 files changed, 88 insertions(+) create mode 100644 .changeset/android-last-character-on-enter.md diff --git a/.changeset/android-last-character-on-enter.md b/.changeset/android-last-character-on-enter.md new file mode 100644 index 00000000..0af4c14a --- /dev/null +++ b/.changeset/android-last-character-on-enter.md @@ -0,0 +1,5 @@ +--- +"aicodeman": patch +--- + +Stop a phone keyboard losing the last character of every message it sends. Android soft keyboards commit the last typed character and send the Enter key in one InputConnection transaction, so the `input` event and the Enter keydown are both processed before any zero-delay timer runs. The orphaned-input recovery from #388 only resolved its candidate on such a timer, and lost it both ways: xterm emits `\r` synchronously from the Enter keydown, so the local-echo composer submitted the prompt before the recovered character existed, and that `\r` bumped the "did xterm speak for this keystroke" counter, so the candidate then stood itself down and dropped the character outright. Pending candidates are now drained synchronously at the next keydown, from xterm's custom key handler — before xterm processes that key — so the counter still holds the value it had while the candidate's own keystroke was current, and the recovered byte reaches the composer ahead of the Enter. Typing on a physical keyboard is unaffected: there, the timer has already resolved the candidate before the next key arrives. diff --git a/src/web/public/terminal-keycode229-recovery.js b/src/web/public/terminal-keycode229-recovery.js index 3e00ea65..d6a95f74 100644 --- a/src/web/public/terminal-keycode229-recovery.js +++ b/src/web/public/terminal-keycode229-recovery.js @@ -66,6 +66,38 @@ let composing = false; const pending = []; + /** + * Resolve every candidate still pending, right now, instead of waiting for + * its zero-delay timer. + * + * Android soft keyboards commit the last character and send the Enter key + * in ONE InputConnection transaction: the `input` event and the Enter + * keydown are both processed before any timer runs. Left on its timer the + * candidate lost BOTH ways — xterm emits '\r' synchronously from the Enter + * keydown (so the local-echo composer submitted the prompt without the + * character), and that '\r' bumps `canonicalCount`, so the candidate then + * read "xterm spoke for this keystroke" and stood down, dropping the + * character outright. That is the "every message loses its last character" + * report from phones. + * + * Draining at the next keydown is correct on both counts: the counter still + * holds the value it had while this candidate's keystroke was current, and + * the byte reaches the composer ahead of whatever the new key emits. + */ + function flushPending() { + for (const candidate of pending.splice(0)) { + if (candidate.timer !== null) { + try { + clearTimer(candidate.timer); + } catch { + // A broken timer host must not break input handling. + } + candidate.timer = null; + } + resolveCandidate(candidate); + } + } + function cancelPending() { for (const candidate of pending.splice(0)) { candidate.active = false; @@ -111,6 +143,11 @@ */ function handleKeyEvent(event) { if (destroyed || event?.type !== 'keydown') return; + // Settle the PREVIOUS keystroke before this one can move the counter or + // reach the PTY — see flushPending(). This runs from xterm's custom key + // handler, i.e. before xterm processes the key, so a recovered character + // is always ordered ahead of the bytes this keydown produces. + flushPending(); keydownSnapshot = canonicalCount; } diff --git a/test/terminal-keycode229-recovery.test.ts b/test/terminal-keycode229-recovery.test.ts index 6a8a38a6..ebddc629 100644 --- a/test/terminal-keycode229-recovery.test.ts +++ b/test/terminal-keycode229-recovery.test.ts @@ -218,6 +218,52 @@ describe('orphaned terminal input recovery', () => { expect(reads).toEqual([]); }); + it('delivers the last character BEFORE the Enter that submits it (defect 4)', () => { + // Android soft keyboards commit the last character and send the Enter key in + // ONE InputConnection transaction, so the `input` event and the Enter keydown + // are processed before any zero-delay timer runs. Two things then went wrong + // with a candidate that only resolved on its timer: + // + // 1. ORDER — xterm emits '\r' synchronously from the Enter keydown, and the + // local-echo composer submits `pendingText` right there. The recovered + // character arrived one macrotask too late to be part of the prompt. + // 2. LOSS — that '\r' bumps the canonical counter, so by the time the + // candidate resolved, `canonicalCount > snapshot` read as "xterm spoke + // for this keystroke" and stood the recovery down. The character was + // dropped outright: every message sent from the phone lost its last + // character. + // + // Resolving pending candidates synchronously at the NEXT keydown fixes both: + // the counter still holds the value it had when that candidate was created, + // and the byte reaches the composer ahead of the Enter. + const h = harness(); + h.keydown(); + h.input('o'); + expect(h.emitted).toEqual([]); + + h.keydown({ key: 'Enter' }); + expect(h.emitted).toEqual(['o']); + + // xterm now emits '\r' for the Enter. The already-resolved candidate must + // not fire a second time when its timer is flushed. + h.controller.notifyCanonicalData(); + h.flushTimers(); + expect(h.emitted).toEqual(['o']); + expect(h.pendingTimers()).toBe(0); + }); + + it('still stands down at the next keydown when xterm spoke for the candidate', () => { + // The synchronous resolve must not become a "forward everything" path: a + // keystroke xterm delivered itself is still a duplicate if recovered. + const h = harness(); + h.keydown(); + h.input('x'); + h.controller.notifyCanonicalData(); + h.keydown({ key: 'Enter' }); + h.flushTimers(); + expect(h.emitted).toEqual([]); + }); + it('ignores input events that are not committed text', () => { const h = harness(); for (const inputType of ['insertCompositionText', 'deleteContentBackward', 'insertLineBreak', 'insertFromPaste']) { From da933d70bedbf501bf689e2eeb57e88d8c324527 Mon Sep 17 00:00:00 2001 From: Michael Grundberg <michael.grundberg@irisity.com> Date: Wed, 16 Sep 2026 08:05:55 +0200 Subject: [PATCH 10/28] feat(sessions): offer to rebuild the sessions a host reboot destroyed A host reboot takes the tmux server down with it, so every pane dies, reconciliation finds nothing to attach to, and the board comes up empty. Picking yesterday's work back up meant finding each conversation in history and resuming it by hand, one at a time. The boot pass now works out what the reboot killed and leaves it on offer. It runs inside restoreMuxSessions(), in the window where reconciliation has reported the dead sessions and cleanupStaleSessions() has not pruned their records yet, which is the only place the records can still be read. The board shows a banner, and nothing is created until the user clicks it. A click rather than an automatic restore is what makes the reboot heuristic acceptable. The heuristic cannot tell a reboot from a crash that took tmux down inside the same window, so it decides whether to ASK, never whether to act: a wrong yes costs a line of text the user dismisses instead of N CLI processes nobody asked for. Four things are re-checked when the click arrives rather than trusted from boot, because hours can pass and the board moves on. The owner's privilege grant re-resolves through the env clamp. The workspace must still be on disk. A conversation the user already resumed by hand from the Resume list is skipped, since two panes running --resume on one conversation would fight over the same transcript. Entries leave the plan synchronously before the first await, and the route is single-flighted, so a double-click or two devices cannot both reach the same entry. A restored session comes back attached, idle and disarmed. Respawn controllers and Ralph loops are deliberately not re-armed: a machine that just came up is the worst moment to turn an autonomous run loose. Its workspace hooks are installed by the restore route itself, because the boot-time sweep sits behind a gate that is false after a reboot and has finished long before the click; without them a session goes silently blind, with no stop or idle events for respawn, no Approvals Inbox item and no red tab on a blocking dialog. Stats collection starts the same way. The pane is new, so the conversation continues and the terminal scrollback does not. The banner says so rather than letting an empty pane read as a broken restore. The plan lives in memory only. A server restart drops it, which costs the convenience this adds and never the conversation: the conversation is the transcript under ~/.claude/projects, which the Welcome screen's Resume list and the Session Manager already read, so a dropped plan returns the user to resuming by hand. clampEnvOverridesForOwner moves to src/session-env-clamp.ts, since the question it answers is about session privilege rather than about HTTP and it now has a caller outside the route layer. Its test hook stays re-exported from session-routes.ts. Claude sessions only for this pass. The other CLIs name their thread in their own config object, which this does not thread through yet. Remote and docker sessions are skipped on purpose, because both need another host or a container to be up and a freshly booted machine cannot promise either. Refs #411 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- .changeset/reboot-restore-banner.md | 5 + src/reboot-restore.ts | 225 +++++++++++++ src/session-env-clamp.ts | 91 ++++++ src/web/public/app.js | 2 + src/web/public/index.html | 19 ++ src/web/public/reboot-restore-ui.js | 95 ++++++ src/web/public/styles.css | 75 +++++ src/web/reboot-restore-registry.ts | 142 +++++++++ src/web/routes/index.ts | 1 + src/web/routes/reboot-restore-routes.ts | 194 ++++++++++++ src/web/routes/session-routes.ts | 67 +--- src/web/schemas.ts | 14 + src/web/server.ts | 57 +++- test/reboot-restore.test.ts | 367 ++++++++++++++++++++++ test/routes/reboot-restore-routes.test.ts | 175 +++++++++++ 15 files changed, 1462 insertions(+), 67 deletions(-) create mode 100644 .changeset/reboot-restore-banner.md create mode 100644 src/reboot-restore.ts create mode 100644 src/session-env-clamp.ts create mode 100644 src/web/public/reboot-restore-ui.js create mode 100644 src/web/reboot-restore-registry.ts create mode 100644 src/web/routes/reboot-restore-routes.ts create mode 100644 test/reboot-restore.test.ts create mode 100644 test/routes/reboot-restore-routes.test.ts diff --git a/.changeset/reboot-restore-banner.md b/.changeset/reboot-restore-banner.md new file mode 100644 index 00000000..33942048 --- /dev/null +++ b/.changeset/reboot-restore-banner.md @@ -0,0 +1,5 @@ +--- +'aicodeman': minor +--- + +Offer to rebuild the sessions a host reboot destroyed. A reboot takes the tmux server down with it, so every pane dies and the board comes up empty. Codeman now works out what was running, and the board offers to restore it behind a click. The conversations come back; the terminal scrollback does not, and the banner says so. diff --git a/src/reboot-restore.ts b/src/reboot-restore.ts new file mode 100644 index 00000000..c8b6bd5e --- /dev/null +++ b/src/reboot-restore.ts @@ -0,0 +1,225 @@ +/** + * @fileoverview Decide which sessions a host reboot destroyed and may be rebuilt. + * + * A server restart and a host reboot both leave `reconcileSessions()` reporting + * dead sessions, and they need opposite handling. A server restart leaves the + * tmux panes running, so recovery ATTACHES to them. A host reboot takes the tmux + * server down with it, so there is nothing to attach to and the pane has to be + * created again. This module holds the decision half of that second case, kept + * free of tmux and disk access so it can be unit tested without either. Every + * observation it reads is gathered by the caller and passed in. + * + * "Eligible" here means a session the user did not end on purpose. The rule that + * an intentional kill or detach is never auto-revived is enforced at runtime by + * an in-memory guard in `TmuxManager`, and memory does not survive a reboot. The + * durable equivalent is the record `cleanupSession()` leaves behind. An unpinned + * kill deletes the record outright, so it is already absent here. A pinned kill + * goes through `demoteOrRemoveSession()` and lands as `status: 'stopped'`, which + * is the marker this module refuses. Pruning keeps a pinned record WITHOUT + * touching its status, so a pinned session a reboot killed still reads `idle` or + * `busy` and stays eligible. + * + * @dependencies types (SessionState), config/cli-registry + * @consumedby web/server (plan build at boot), web/routes/reboot-restore-routes + * + * @module reboot-restore + */ + +import type { SessionState } from './types.js'; +import { getCli } from './config/cli-registry/registry.js'; + +/** Session statuses a reboot restore may rebuild. `stopped` is the kill marker. */ +const RESTORABLE_STATUSES: ReadonlySet<string> = new Set(['idle', 'busy', 'error']); + +/** Observations the reboot heuristic reads. Gathered by the caller, never here. */ +export interface RebootEvidence { + /** Sessions that still had a live pane during reconciliation. */ + livePaneCount: number; + /** Sessions reconciliation just marked dead. */ + deadSessionCount: number; + /** `os.uptime()`, in seconds. */ + uptimeSeconds: number; + /** Newest `lastActivityAt` across the persisted records, in ms since the epoch. */ + newestPersistedActivityAt: number; + /** `Date.now()` when the evidence was gathered, in ms. */ + now: number; +} + +/** + * Decide whether the machine plausibly rebooted rather than the server restarting. + * + * Two signals have to agree. The socket must hold no panes at all while state + * still lists sessions, which rules out an ordinary server restart. The host + * must also have booted after the newest persisted session activity, which is + * the corroboration `os.uptime()` provides cheaply. A wiped tmux socket on a + * long-uptime host fails the second test, so a user who killed the tmux server + * by hand does not get every session offered back to them. + * + * This heuristic decides whether to ASK, never whether to act. A wrong yes costs + * the user a banner they dismiss, because the restore itself waits for a click. + */ +export function looksLikeHostReboot(evidence: RebootEvidence): boolean { + if (evidence.deadSessionCount === 0) return false; + if (evidence.livePaneCount > 0) return false; + if (evidence.newestPersistedActivityAt <= 0) return false; + const bootedAt = evidence.now - evidence.uptimeSeconds * 1000; + return bootedAt > evidence.newestPersistedActivityAt; +} + +/** + * Pick the conversation the rebuilt pane should resume. + * + * The chain's tail is the newest conversation the session was holding, which is + * what a compact or a clear leaves behind; `resumeSessionId` covers a session + * that was itself started as a resume, and the session id is the original + * conversation for everything else. + */ +export function resolveResumeConversationId(state: SessionState): string { + const chain = state.claudeSessionChain; + const chainTail = Array.isArray(chain) && chain.length > 0 ? chain[chain.length - 1] : undefined; + return chainTail || state.resumeSessionId || state.id; +} + +/** Why one session was passed over. Reported for logging and assertions. */ +export interface RebootRestoreRejection { + sessionId: string; + reason: + | 'no-persisted-record' + | 'intentionally-ended' + | 'respawn-blocked' + | 'remote-or-docker' + | 'unsupported-mode' + | 'no-working-dir' + | 'workspace-missing' + | 'already-live'; +} + +/** One restorable session, as the banner shows it and the rebuild replays it. */ +export interface RebootRestoreEntry { + sessionId: string; + name?: string; + workingDir: string; + owner?: string; + mode: string; + /** The conversation the rebuilt pane resumes. */ + resumeConversationId: string; + /** + * The persisted record, kept whole so the rebuild can replay what it held. + * Read at boot, before pruning deletes it, and held in memory until the click. + */ + state: SessionState; +} + +export interface RebootRestorePlan { + restore: RebootRestoreEntry[]; + skipped: RebootRestoreRejection[]; +} + +/** + * Split the sessions reconciliation just killed into the ones a reboot restore + * may offer and the ones it must leave alone. + * + * @param deadSessionIds Session ids `reconcileSessions()` reported as dead. + * @param persisted The `state.json` session records, which `cleanupStaleSessions()` + * has not pruned yet at the point this runs. + * @param workspaceExists Whether a working directory is still on disk. A tmux + * session can outlive its deleted repo, and rebuilding one there would scaffold + * an empty tree. The caller owns the disk access; the click re-checks, because + * a repo can be deleted between the boot and the click. + */ +export function planRebootRestore( + deadSessionIds: readonly string[], + persisted: Readonly<Record<string, SessionState>>, + workspaceExists: (workingDir: string) => boolean +): RebootRestorePlan { + const restore: RebootRestoreEntry[] = []; + const skipped: RebootRestoreRejection[] = []; + + for (const sessionId of deadSessionIds) { + const state = persisted[sessionId]; + if (!state) { + // An unpinned kill already deleted the record, so absence IS the guard. + skipped.push({ sessionId, reason: 'no-persisted-record' }); + continue; + } + if (!RESTORABLE_STATUSES.has(state.status)) { + // A pinned kill was demoted to `stopped`. Reviving it would undo the kill. + skipped.push({ sessionId, reason: 'intentionally-ended' }); + continue; + } + if (state.respawnBlocked === true) { + // The crash-loop breaker tripped on this pane. Re-creating it restarts the loop. + skipped.push({ sessionId, reason: 'respawn-blocked' }); + continue; + } + if (state.remote || state.docker) { + // Both need another host or a container to be up, which a just-booted machine + // cannot promise. The remote reconnect watcher owns the remote case already. + skipped.push({ sessionId, reason: 'remote-or-docker' }); + continue; + } + // Capability, not a CLI id: this pass resumes by handing the CLI a conversation + // id through the top-level `resumeSessionId`, which only a CLI whose history the + // claude-jsonl reader understands can consume that way. Others carry their thread + // id in their own `<Mode>Config`, which this pass does not thread through. + if (getCli(state.mode ?? 'claude')?.capabilities.transcript !== 'claude-jsonl') { + skipped.push({ sessionId, reason: 'unsupported-mode' }); + continue; + } + if (!state.workingDir) { + skipped.push({ sessionId, reason: 'no-working-dir' }); + continue; + } + if (!workspaceExists(state.workingDir)) { + skipped.push({ sessionId, reason: 'workspace-missing' }); + continue; + } + restore.push({ + sessionId, + name: state.name, + workingDir: state.workingDir, + owner: state.owner, + mode: state.mode ?? 'claude', + resumeConversationId: resolveResumeConversationId(state), + state, + }); + } + + return { restore, skipped }; +} + +/** + * Drop the entries whose conversation is already on screen. + * + * Hours can pass between the boot that built the plan and the click that spends + * it, and the Resume list can reach the same conversation in the meantime. Two + * panes running `claude --resume` on one conversation is the failure this + * prevents, so a match on either the session id or the conversation id is enough + * to skip the entry. + */ +export function rejectAlreadyLive( + entries: readonly RebootRestoreEntry[], + liveSessionIds: ReadonlySet<string>, + liveConversationIds: ReadonlySet<string> +): RebootRestorePlan { + const restore: RebootRestoreEntry[] = []; + const skipped: RebootRestoreRejection[] = []; + for (const entry of entries) { + if (liveSessionIds.has(entry.sessionId) || liveConversationIds.has(entry.resumeConversationId)) { + skipped.push({ sessionId: entry.sessionId, reason: 'already-live' }); + continue; + } + restore.push(entry); + } + return { restore, skipped }; +} + +/** Newest `lastActivityAt` across persisted records, or 0 when there are none. */ +export function newestPersistedActivity(persisted: Readonly<Record<string, SessionState>>): number { + let newest = 0; + for (const state of Object.values(persisted)) { + const stamp = state.lastActivityAt ?? state.createdAt ?? 0; + if (stamp > newest) newest = stamp; + } + return newest; +} diff --git a/src/session-env-clamp.ts b/src/session-env-clamp.ts new file mode 100644 index 00000000..edaecd5c --- /dev/null +++ b/src/session-env-clamp.ts @@ -0,0 +1,91 @@ +/** + * @fileoverview The env-var half of the multi-user privilege clamp. + * + * A session's `envOverrides` can hand back privilege that the per-CLI config + * clamp removed, so a non-granted owner's overrides get the privileged keys + * stripped before the session is built. Two callers need that today. The create + * and resume routes clamp what a request asked for, and the reboot-restore route + * clamps what a persisted record carried, because a record written while its + * owner held a grant must not replay that grant after the grant is gone. + * + * This lives outside `web/routes` on purpose. The question it answers is about + * session privilege rather than about HTTP, and `cron/cron-service.ts` sets the + * precedent by importing `canUsernameRunPrivilegedCommands` from `user-store.ts` + * directly and re-resolving the owner's grant when a job fires. Every caller here + * re-resolves the grant at the moment it builds a session, for the same reason. + * + * @dependencies user-store (canUsernameRunPrivilegedCommands), config/cli-registry + * @consumedby web/routes/session-routes, web/routes/reboot-restore-routes + * + * @module session-env-clamp + */ + +import { canUsernameRunPrivilegedCommands } from './user-store.js'; +import { enabledClis } from './config/cli-registry/registry.js'; + +/** + * Env-var keys a non-granted owner must not be able to set, because each one + * hands back privilege `clampExternalCliBypassForOwner()` just removed, or redirects a + * credential-resolution endpoint. + * + * The DeepSeek three are reachable because `DSH_*` and `DEEPSEEK_*` are + * allowlisted `envOverrides` prefixes (schemas.ts) — which they have to be, since + * that is also how a user configures the harness's non-privileged knobs. + * + * - `DSH_PERMISSION_MODE` IS the harness's permission switch. Every other CLI's + * bypass is a command-line FLAG, reachable only through the per-CLI config the + * clamp already owns; this one is an env var, so the config clamp alone is + * half a gate. + * - `DSH_HOME` points the launcher at a profile tree, and a profile's plugin code + * executes at BOOT, before any approval row can apply. A user who can write a + * workspace can put a profile in it, so this is the wider of the two. + * - `DEEPSEEK_BASE_URL` aims the provider endpoint, and `_configureCliEnv()` + * forwards the SERVER's own `DEEPSEEK_API_KEY` into every dsh pane before + * `applyEnvOverrides()` runs — so a non-granted owner who could set the base + * URL would have the operator's API key sent as a bearer credential to a host + * of their choosing. (`DEEPSEEK_API_KEY` itself stays overridable: supplying + * your OWN key removes privilege rather than granting it.) + * - `OMP_AUTH_BROKER_URL`/`OMP_AUTH_BROKER_TOKEN` are where omp resolves + * credentials from — the same shape as `DEEPSEEK_BASE_URL` above, reachable + * because `OMP_*` is an allowlisted prefix. Unlike DeepSeek, Codeman does not + * forward any operator-held key into an omp pane today (omp's provider + * credentials live in `~/.omp` config files, not env vars), so there is no + * known concrete exfiltration path yet — clamped defensively anyway, since a + * non-granted owner redirecting where a shared multi-tenant deployment + * resolves auth from is not something to allow silently (found in + * Ark0N/Codeman#353 review; omp's own knobs are otherwise mostly `PI_*`, + * already allowlisted for pi and not addressed here — see resolveOmpHome()). + */ +export function ownerClampedEnvKeys(): string[] { + return enabledClis().flatMap((entry) => entry.capabilities.privilegedEnvKeys); +} + +/** + * Env-var half of the multi-user bypass clamp. + * + * `clampExternalCliBypassForOwner()` in `web/routes/session-routes.ts` clamps the + * per-CLI CONFIG, and for every CLI + * but DeepSeek that is the whole story. Here it is not: `applyEnvOverrides()` runs + * AFTER `_configureCliEnv()` in tmux-manager, so an override sent on the SAME + * request lands last and wins, and a non-granted owner could restore + * `danger-full-access` on the very request the config clamp downgraded. + * + * Keys are DROPPED rather than rewritten: dropping falls through to what + * `_configureCliEnv()` exports, which is the clamped config and the server's own + * `DSH_HOME`, i.e. exactly the intended state. No-op in single-user mode and for a + * granted owner, like every other clamp here + * (`canUsernameRunPrivilegedCommands()` returns true when `!isMultiUserMode()`), + * and it returns the caller's own object untouched when there is nothing to strip. + */ +export async function clampEnvOverridesForOwner( + owner: string | undefined, + envOverrides: Record<string, string> | undefined +): Promise<Record<string, string> | undefined> { + if (!envOverrides) return envOverrides; + const keys = ownerClampedEnvKeys(); + if (!keys.some((key) => key in envOverrides)) return envOverrides; + if (await canUsernameRunPrivilegedCommands(owner)) return envOverrides; + const clamped = { ...envOverrides }; + for (const key of keys) delete clamped[key]; + return clamped; +} diff --git a/src/web/public/app.js b/src/web/public/app.js index 4c688899..1f0ae409 100644 --- a/src/web/public/app.js +++ b/src/web/public/app.js @@ -957,6 +957,8 @@ class CodemanApp { this.registerServiceWorker(); // Fetch tunnel status for header indicator (desktop only) this.loadTunnelStatus(); + // Ask whether a host reboot left sessions worth rebuilding (banner, never automatic) + this.initRebootRestoreBanner?.(); // Share a single settings fetch between both consumers const settingsPromise = fetch('/api/settings').then(r => r.ok ? r.json() : null).then(env => env?.data ?? null).catch(() => null); this.loadQuickStartCases(null, settingsPromise); diff --git a/src/web/public/index.html b/src/web/public/index.html index 2ba80243..cbe9a480 100644 --- a/src/web/public/index.html +++ b/src/web/public/index.html @@ -213,6 +213,24 @@ <button class="offline-banner-retry" id="offlineBannerRetry" onclick="app.retryConnection()">Retry now</button> </div> + <!-- Reboot-restore offer: shown when the server found sessions a host reboot + killed and is asking whether to rebuild them. Populated by + reboot-restore-ui.js; nothing is created until the user clicks. --> + <div class="reboot-restore-banner" id="rebootRestoreBanner" role="status" hidden> + <span class="reboot-restore-banner-icon" aria-hidden="true">↺</span> + <span class="reboot-restore-banner-text" id="rebootRestoreBannerText"></span> + <span class="reboot-restore-banner-detail" id="rebootRestoreBannerDetail"></span> + <span class="reboot-restore-banner-note">Conversations return; terminal history does not.</span> + <button + class="reboot-restore-banner-accept" + id="rebootRestoreBannerAccept" + onclick="app.restoreRebootSessions()" + > + Restore + </button> + <button class="reboot-restore-banner-dismiss" onclick="app.dismissRebootRestore()">Dismiss</button> + </div> + <!-- Timer Banner (shown when timed run is active) --> <div class="timer-banner" id="timerBanner" style="display: none;"> <div class="timer-content"> @@ -3535,6 +3553,7 @@ <script defer src="readmymind-ui.js"></script> <script defer src="ultracode-panel.js"></script> <script defer src="approvals-ui.js"></script> + <script defer src="reboot-restore-ui.js"></script> <script defer src="admin-ui.js"></script> <script defer src="session-ui.js"></script> <script defer src="webview-tabs.js"></script> diff --git a/src/web/public/reboot-restore-ui.js b/src/web/public/reboot-restore-ui.js new file mode 100644 index 00000000..0869eebe --- /dev/null +++ b/src/web/public/reboot-restore-ui.js @@ -0,0 +1,95 @@ +/** + * @fileoverview Reboot-restore banner: offer back the sessions a host reboot destroyed. + * + * A host reboot takes the tmux server down with it, so every session's pane dies + * and the board comes up empty. The server works out what was running from the + * records it still holds at boot, and this banner asks the user whether to + * rebuild them. Nothing is created until they click, because the server's + * reboot guess is a heuristic and a wrong automatic restore would spawn CLI + * processes nobody asked for. + * + * Seeded once from `GET /api/reboot-restore` on init. Restore posts to + * `POST /api/reboot-restore/restore`, Dismiss posts to + * `POST /api/reboot-restore/dismiss`, and either way the banner goes away. The + * restored sessions arrive as ordinary `session:created` events, so no extra + * rendering is needed here. + * + * The banner says that terminal history did not survive, because a restored + * session is a new pane: the conversation continues and the scrollback does not. + * Saying so is what keeps an empty pane from reading as a broken restore. + * Backend: src/web/reboot-restore-registry.ts, src/web/routes/reboot-restore-routes.ts. + * + * @mixin Extends CodemanApp.prototype via Object.assign + * @dependency app.js (CodemanApp class, showToast) + * @dependency api-client.js at runtime (this._apiJson / this._apiPost) + * @loadorder 11.7 of 17, after approvals-ui.js + */ + +Object.assign(CodemanApp.prototype, { + /** Ask the server whether a reboot left anything on offer, and show the banner if so. */ + async initRebootRestoreBanner() { + const data = await this._apiJson('/api/reboot-restore'); + const sessions = data?.sessions ?? []; + if (sessions.length === 0) return; + this._rebootRestoreSessions = sessions; + this.renderRebootRestoreBanner(); + }, + + renderRebootRestoreBanner() { + const banner = this.$('rebootRestoreBanner'); + if (!banner) return; + const sessions = this._rebootRestoreSessions ?? []; + if (sessions.length === 0) { + banner.hidden = true; + return; + } + const count = sessions.length; + const text = this.$('rebootRestoreBannerText'); + if (text) { + const noun = count === 1 ? 'session' : 'sessions'; + text.textContent = `Restore ${count} ${noun} from before the reboot`; + } + const detail = this.$('rebootRestoreBannerDetail'); + if (detail) { + // Names, so the user can tell what they are about to relaunch. + const names = sessions + .map((s) => s.name || s.workingDir?.split('/').pop() || s.id.slice(0, 8)) + .slice(0, 4) + .join(', '); + detail.textContent = count > 4 ? `${names}, …` : names; + detail.title = sessions.map((s) => `${s.name || s.id}\n${s.workingDir}`).join('\n\n'); + } + banner.hidden = false; + }, + + /** Rebuild everything on offer. The panes are new, so scrollback does not come back. */ + async restoreRebootSessions() { + const button = this.$('rebootRestoreBannerAccept'); + if (button) button.disabled = true; + const res = await this._apiPost('/api/reboot-restore/restore', {}); + const body = res && res.ok ? await res.json().catch(() => null) : null; + if (!body) { + if (button) button.disabled = false; + this.showToast?.('Could not restore the sessions', 'error'); + return; + } + const restored = body.restored?.length ?? 0; + const skipped = body.skipped?.length ?? 0; + this._rebootRestoreSessions = []; + this.renderRebootRestoreBanner(); + if (restored > 0) { + const noun = restored === 1 ? 'conversation' : 'conversations'; + this.showToast?.(`Restored ${restored} ${noun}. Terminal history did not survive the reboot.`, 'success'); + } + if (skipped > 0) { + this.showToast?.(`${skipped} could not be restored (workspace gone, or already open)`, 'warning'); + } + }, + + /** Drop the offer. The Resume list still reaches every one of these conversations. */ + async dismissRebootRestore() { + this._rebootRestoreSessions = []; + this.renderRebootRestoreBanner(); + await this._apiPost('/api/reboot-restore/dismiss', {}); + }, +}); diff --git a/src/web/public/styles.css b/src/web/public/styles.css index 398eb6ce..b5e4873b 100644 --- a/src/web/public/styles.css +++ b/src/web/public/styles.css @@ -15243,6 +15243,81 @@ html[data-skin="daylight-blue"] .welcome-btn-tunnel.active:hover { skin, including the light ones. Visibility is driven by the `hidden` attribute, so the display rules need !important to lose to it. */ +/* Reboot-restore offer. Amber rather than red: nothing is wrong, the board is + asking a question, and the user can ignore it. See reboot-restore-ui.js. */ +.reboot-restore-banner { + display: flex; + align-items: center; + gap: 0.6rem; + padding: 0.45rem 1rem; + background: linear-gradient(90deg, #b45309, #92400e); + border-bottom: 1px solid rgba(0, 0, 0, 0.35); + color: #fff; + font-size: 0.78rem; + font-weight: 600; + letter-spacing: 0.01em; + flex-shrink: 0; + z-index: 1250; +} + +.reboot-restore-banner[hidden] { + display: none !important; +} + +.reboot-restore-banner-icon { + flex-shrink: 0; + font-size: 0.95rem; + line-height: 1; +} + +.reboot-restore-banner-text { + white-space: nowrap; +} + +.reboot-restore-banner-detail { + color: rgba(255, 255, 255, 0.8); + font-weight: 500; + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; +} + +.reboot-restore-banner-note { + color: rgba(255, 255, 255, 0.75); + font-weight: 500; + white-space: nowrap; + margin-left: auto; +} + +.reboot-restore-banner-accept, +.reboot-restore-banner-dismiss { + flex-shrink: 0; + padding: 0.2rem 0.6rem; + border-radius: 5px; + border: 1px solid rgba(255, 255, 255, 0.55); + background: rgba(255, 255, 255, 0.12); + color: #fff; + font-size: 0.72rem; + font-weight: 600; + cursor: pointer; +} + +.reboot-restore-banner-accept:hover, +.reboot-restore-banner-dismiss:hover { + background: rgba(255, 255, 255, 0.24); +} + +.reboot-restore-banner-accept:disabled { + opacity: 0.6; + cursor: default; +} + +.reboot-restore-banner-dismiss { + border-color: rgba(255, 255, 255, 0.3); + background: transparent; + font-weight: 500; +} + .offline-banner { display: flex; align-items: center; diff --git a/src/web/reboot-restore-registry.ts b/src/web/reboot-restore-registry.ts new file mode 100644 index 00000000..e265fa62 --- /dev/null +++ b/src/web/reboot-restore-registry.ts @@ -0,0 +1,142 @@ +/** + * @fileoverview The pending restore plan: what a host reboot destroyed, waiting on a click. + * + * The boot pass builds this plan inside `restoreMuxSessions()`, in the window + * where reconciliation has reported the dead sessions and `cleanupStaleSessions()` + * has not pruned their records yet. The board then offers "restore N sessions + * from before the reboot", and `web/routes/reboot-restore-routes` spends the plan + * when the user clicks. + * + * Invariants: + * - Entries are in-memory only. A server restart drops the plan, and nothing + * re-builds it, because the records it was built from are pruned by then. + * That costs the convenience this feature adds and never the conversation: + * the conversation IS the transcript under `~/.claude/projects`, which + * `services/unified-session-service.ts` reads for the Welcome screen's Resume + * list and the Session Manager, and `resumeHistorySession()` in + * `web/public/terminal-ui.js` resumes from a row there with no persisted + * session record involved. A dropped plan therefore returns the user to + * resuming by hand, one at a time, which is where they are without this + * feature. What the plan held that a transcript does not is the owner, the + * name, the env overrides, the effort and the lineage. + * - Module-level singleton in the style of `web/approval-inbox.ts`: no `Session` + * import and no IO, which keeps it unit-testable and cycle-free. + * - Spending is take-then-build: `take()` removes entries synchronously, before + * the route's first `await`, so a double-click or two devices cannot both + * reach the same entry and put two panes on one conversation. + * - One restore runs at a time. `beginSpending()` single-flights the route, so + * two concurrent clicks cannot interleave pane creation. + * + * @dependencies reboot-restore (RebootRestoreEntry) + * @consumedby web/server (plan build at boot), web/routes/reboot-restore-routes + * + * @module web/reboot-restore-registry + */ + +import type { RebootRestoreEntry } from '../reboot-restore.js'; + +/** + * A plan older than this is dropped on read. A machine that rebooted yesterday + * has moved on, and an offer nobody took by then is noise rather than a rescue. + */ +const PLAN_TTL_MS = 24 * 60 * 60 * 1000; + +export class RebootRestoreRegistry { + /** Keyed by session id, in the order the boot pass found them. */ + private entries = new Map<string, RebootRestoreEntry>(); + /** When the boot pass built the plan, in ms since the epoch. */ + private builtAt = 0; + /** True while a restore route call is between its take and its last pane. */ + private spending = false; + + /** Replace the plan with what the boot pass found. An empty list clears it. */ + set(entries: readonly RebootRestoreEntry[]): void { + this.entries = new Map(entries.map((entry) => [entry.sessionId, entry])); + this.builtAt = entries.length > 0 ? Date.now() : 0; + } + + /** + * The entries a viewer may see, newest plan first-come order preserved. + * + * @param canAccess Ownership predicate, so a user sees their own entries and + * an admin sees all. Applied here rather than in the route so the count the + * banner shows and the entries a click spends come from one filter. + */ + list(canAccess: (owner: string | undefined) => boolean): RebootRestoreEntry[] { + this.dropIfExpired(); + return [...this.entries.values()].filter((entry) => canAccess(entry.owner)); + } + + /** + * Remove and return the entries a click is about to spend. + * + * Synchronous and total: an entry leaves the plan here, before any pane is + * created, so a second click finds nothing to spend. Entries a caller may not + * access are left in place, and unknown ids are ignored. + * + * @param sessionIds The ids to spend, or undefined for every visible entry. + */ + take(canAccess: (owner: string | undefined) => boolean, sessionIds?: readonly string[]): RebootRestoreEntry[] { + this.dropIfExpired(); + const wanted = sessionIds ? new Set(sessionIds) : undefined; + const taken: RebootRestoreEntry[] = []; + for (const entry of [...this.entries.values()]) { + if (wanted && !wanted.has(entry.sessionId)) continue; + if (!canAccess(entry.owner)) continue; + this.entries.delete(entry.sessionId); + taken.push(entry); + } + return taken; + } + + /** + * Put entries back after a rebuild never got as far as creating a pane. + * + * Used for the click-time rejections, so a conversation the user resumed by + * hand meanwhile does not silently vanish from the banner while a workspace + * that came back stays offered. + */ + restore(entries: readonly RebootRestoreEntry[]): void { + for (const entry of entries) this.entries.set(entry.sessionId, entry); + if (entries.length > 0 && this.builtAt === 0) this.builtAt = Date.now(); + } + + /** Drop the entries a viewer can see. Returns how many went. */ + clear(canAccess: (owner: string | undefined) => boolean): number { + const removable = [...this.entries.values()].filter((entry) => canAccess(entry.owner)); + for (const entry of removable) this.entries.delete(entry.sessionId); + if (this.entries.size === 0) this.builtAt = 0; + return removable.length; + } + + /** + * Claim the right to run a restore, or report that one is already running. + * Callers that get `true` must call `endSpending()` in a `finally`. + */ + beginSpending(): boolean { + if (this.spending) return false; + this.spending = true; + return true; + } + + endSpending(): void { + this.spending = false; + } + + /** Test hook: forget everything, including the single-flight claim. */ + reset(): void { + this.entries.clear(); + this.builtAt = 0; + this.spending = false; + } + + private dropIfExpired(): void { + if (this.builtAt > 0 && Date.now() - this.builtAt > PLAN_TTL_MS) { + this.entries.clear(); + this.builtAt = 0; + } + } +} + +/** Process-wide singleton, mirroring `approvalInbox`. */ +export const rebootRestoreRegistry = new RebootRestoreRegistry(); diff --git a/src/web/routes/index.ts b/src/web/routes/index.ts index c1f0a952..f11d6194 100644 --- a/src/web/routes/index.ts +++ b/src/web/routes/index.ts @@ -11,6 +11,7 @@ export { registerCronRoutes } from './cron-routes.js'; export { registerSystemRoutes } from './system-routes.js'; export { registerHookEventRoutes } from './hook-event-routes.js'; export { registerApprovalRoutes } from './approval-routes.js'; +export { registerRebootRestoreRoutes } from './reboot-restore-routes.js'; export { registerReadMyMindRoutes } from './readmymind-routes.js'; export { registerStatusTelemetryRoutes } from './status-telemetry-routes.js'; export { registerCaseRoutes } from './case-routes.js'; diff --git a/src/web/routes/reboot-restore-routes.ts b/src/web/routes/reboot-restore-routes.ts new file mode 100644 index 00000000..580b17c6 --- /dev/null +++ b/src/web/routes/reboot-restore-routes.ts @@ -0,0 +1,194 @@ +/** + * @fileoverview Reboot-restore routes: offer back the sessions a host reboot destroyed. + * + * The boot pass leaves a plan in `web/reboot-restore-registry` when the machine + * plausibly rebooted. The board reads it, shows a banner, and the user decides: + * - `GET /api/reboot-restore`: what is on offer, ownership-scoped + * - `POST /api/reboot-restore/restore`: rebuild some or all of it + * - `POST /api/reboot-restore/dismiss`: drop the offer + * + * A click, not the heuristic, is what creates panes. The heuristic only decides + * whether the banner appears, so a wrong yes costs a line of text the user + * dismisses rather than N CLI processes nobody asked for. + * + * Rebuilding is take-then-build: entries leave the plan synchronously at the top + * of the route, before the first `await`, and the whole route is single-flighted, + * so a double-click or two devices cannot put two panes on one conversation. + * Three things are re-checked at click time rather than trusted from boot: the + * owner's privilege grant, the workspace still being on disk, and the + * conversation not already being live because the user resumed it by hand. + * + * A rebuilt session comes back attached, idle and disarmed. Respawn controllers + * and Ralph loops are deliberately not re-armed, and its terminal scrollback is + * gone, because the pane is new. The banner says so. + */ + +import { FastifyInstance } from 'fastify'; +import { existsSync } from 'node:fs'; +import { ApiErrorCode, createErrorResponse, getErrorMessage } from '../../types.js'; +import { RebootRestoreRequestSchema } from '../schemas.js'; +import { parseBody, getAuthUser, canAccessOwned } from '../route-helpers.js'; +import { rebootRestoreRegistry } from '../reboot-restore-registry.js'; +import { rejectAlreadyLive, type RebootRestoreEntry, type RebootRestoreRejection } from '../../reboot-restore.js'; +import { clampEnvOverridesForOwner } from '../../session-env-clamp.js'; +import { Session } from '../../session.js'; +import { resolveClaudeModeForUsername } from '../../user-store.js'; +import { getCli } from '../../config/cli-registry/registry.js'; +import { applyWorkspaceHooks } from '../../hooks-config.js'; +import { getLifecycleLog } from '../../session-lifecycle-log.js'; +import { STATS_COLLECTION_INTERVAL_MS } from '../../config/server-timing.js'; +import { SseEvent } from '../sse-events.js'; +import type { SessionAttachmentHistoryItem } from '../../types.js'; +import type { SessionPort, EventPort, ConfigPort, InfraPort } from '../ports/index.js'; + +type RebootRestoreCtx = SessionPort & EventPort & ConfigPort & InfraPort; + +/** The banner's view of one restorable session. The record itself never leaves the server. */ +function toBannerItem(entry: RebootRestoreEntry) { + return { + id: entry.sessionId, + name: entry.name, + workingDir: entry.workingDir, + mode: entry.mode, + owner: entry.owner, + }; +} + +export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRestoreCtx): void { + const accessorFor = (req: Parameters<typeof getAuthUser>[0]) => { + const user = getAuthUser(req); + return (owner: string | undefined) => canAccessOwned(user, owner); + }; + + // ========== What is on offer ========== + + app.get('/api/reboot-restore', async (req) => { + const entries = rebootRestoreRegistry.list(accessorFor(req)); + return { + sessions: entries.map(toBannerItem), + // Said plainly here so the banner never implies a full restore: the pane is + // new, so the conversation continues and the terminal history does not. + scrollbackRestored: false, + }; + }); + + // ========== Spend it ========== + + app.post('/api/reboot-restore/restore', async (req, reply) => { + const body = parseBody(RebootRestoreRequestSchema, req.body, 'Invalid reboot restore request'); + const canAccess = accessorFor(req); + + // Take BEFORE the first await: a second click must find nothing to spend. + if (!rebootRestoreRegistry.beginSpending()) { + return reply.code(409).send(createErrorResponse(ApiErrorCode.CONFLICT, 'A reboot restore is already running')); + } + const taken = rebootRestoreRegistry.take(canAccess, body.sessionIds); + + try { + if (taken.length === 0) return { restored: [], skipped: [] }; + + // The plan was built at boot and the board has moved on since. A conversation + // the user resumed by hand from the Resume list is already on screen, and a + // second pane on it would fight the first for the same transcript. + const liveSessionIds = new Set(ctx.sessions.keys()); + const liveConversationIds = new Set( + [...ctx.sessions.values()].map((session) => session.claudeSessionId).filter((id): id is string => !!id) + ); + const { restore, skipped } = rejectAlreadyLive(taken, liveSessionIds, liveConversationIds); + // An entry nothing rebuilt stays on offer rather than disappearing silently. + rebootRestoreRegistry.restore(skipped.map((s) => taken.find((e) => e.sessionId === s.sessionId)!)); + + const restored: ReturnType<typeof toBannerItem>[] = []; + const failures: RebootRestoreRejection[] = [...skipped]; + const workspaceHooksEnabled = await ctx.getWorkspaceHooksEnabled(); + + for (const entry of restore) { + // A repo can be deleted between the boot that planned this and the click. + if (!existsSync(entry.workingDir)) { + failures.push({ sessionId: entry.sessionId, reason: 'workspace-missing' }); + continue; + } + try { + const saved = entry.state; + const claudeModeConfig = await ctx.getClaudeModeConfig(); + const session = new Session({ + // The old id is reused on purpose: a pinned record, subagent parents, + // window states and the lifecycle log all key off it, and the unpinned + // record is gone, so there is nothing to collide with. + id: saved.id, + workingDir: saved.workingDir, + mode: saved.mode, + name: saved.name, + createdAt: saved.createdAt, + mux: ctx.mux, + useMux: true, + // No `muxSession`: the reboot took the pane with it, so `startInteractive()` + // takes its create branch and makes a fresh one. + claudeMode: await resolveClaudeModeForUsername(claudeModeConfig.claudeMode, saved.owner), + allowedTools: claudeModeConfig.allowedTools, + resumeSessionId: entry.resumeConversationId, + // Re-resolved against the owner's CURRENT grant, never replayed from the + // record: a grant held when the record was written may be gone now. + envOverrides: await clampEnvOverridesForOwner( + saved.owner, + (saved as { __envOverrides?: Record<string, string> }).__envOverrides + ), + effort: saved.effort, + attachmentHistory: + (saved as { __attachmentHistory?: SessionAttachmentHistoryItem[] }).__attachmentHistory ?? + saved.attachmentHistory, + lastSubmitAt: saved.lastSubmitAt, + claudeSessionChain: saved.claudeSessionChain, + lastActivityAt: saved.lastActivityAt, + owner: saved.owner, + parentSessionId: saved.parentSessionId, + }); + + await ctx.addSession(session); + ctx.persistSessionState(session); + await ctx.setupSessionListeners(session); + await session.startInteractive(); + + // A session without its workspace hooks goes silently blind: no stop or + // idle events for respawn, no Approvals Inbox item, no red tab on a + // blocking dialog. The boot-time sweep finished hours ago, so the click + // path installs them itself. `hooks: 'always'` is the capability that says + // this CLI installs Codeman's hooks into the workspace. + if (workspaceHooksEnabled && getCli(session.mode)?.capabilities.hooks === 'always') { + await applyWorkspaceHooks(session.workingDir, true).catch((err: unknown) => + console.warn(`[reboot-restore] hook install failed for ${session.workingDir}: ${getErrorMessage(err)}`) + ); + } + + getLifecycleLog().log({ event: 'recovered', sessionId: session.id, name: session.name }); + // Every other open tab and phone needs this; the clicking tab already has + // the response, and the client's handler is an idempotent upsert. + ctx.broadcast(SseEvent.SessionCreated, ctx.getSessionStateWithRespawn(session)); + restored.push(toBannerItem(entry)); + } catch (err) { + // One workspace that has gone missing must not stop the rest of the pass. + console.error(`[reboot-restore] failed to rebuild ${entry.sessionId}:`, err); + failures.push({ sessionId: entry.sessionId, reason: 'workspace-missing' }); + } + } + + if (restored.length > 0) { + // A reboot leaves recovery with nothing alive to find, so its own block never + // started the stats collector. This clears and re-arms its interval, so it is + // safe to call whether or not the collector is already running. + ctx.mux.startStatsCollection(STATS_COLLECTION_INTERVAL_MS); + } + + return { restored, skipped: failures }; + } finally { + rebootRestoreRegistry.endSpending(); + } + }); + + // ========== Drop it ========== + + app.post('/api/reboot-restore/dismiss', async (req) => { + const dismissed = rebootRestoreRegistry.clear(accessorFor(req)); + return { dismissed }; + }); +} diff --git a/src/web/routes/session-routes.ts b/src/web/routes/session-routes.ts index 03f92e5d..607ec540 100644 --- a/src/web/routes/session-routes.ts +++ b/src/web/routes/session-routes.ts @@ -89,6 +89,7 @@ import { } from '../route-helpers.js'; import { buildAgentCaseMarker, writeAgentCaseMarker } from '../../agent-case-marker.js'; import { canUsernameRunPrivilegedCommands, resolveClaudeModeForUsername } from '../../user-store.js'; +import { clampEnvOverridesForOwner } from '../../session-env-clamp.js'; import { enabledClis, getCli } from '../../config/cli-registry/registry.js'; import { resolveCliLaunchError } from '../../utils/cli-launcher.js'; import { legacyConfigForMode } from '../../session-cli-registry-bridge.js'; @@ -442,72 +443,6 @@ export async function _clampExternalCliBypassForOwner( }; } -/** - * Env-var keys a non-granted owner must not be able to set, because each one - * hands back privilege the config clamp above just removed, or redirects a - * credential-resolution endpoint. - * - * The DeepSeek three are reachable because `DSH_*` and `DEEPSEEK_*` are - * allowlisted `envOverrides` prefixes (schemas.ts) — which they have to be, since - * that is also how a user configures the harness's non-privileged knobs. - * - * - `DSH_PERMISSION_MODE` IS the harness's permission switch. Every other CLI's - * bypass is a command-line FLAG, reachable only through the per-CLI config the - * clamp already owns; this one is an env var, so the config clamp alone is - * half a gate. - * - `DSH_HOME` points the launcher at a profile tree, and a profile's plugin code - * executes at BOOT, before any approval row can apply. A user who can write a - * workspace can put a profile in it, so this is the wider of the two. - * - `DEEPSEEK_BASE_URL` aims the provider endpoint, and `_configureCliEnv()` - * forwards the SERVER's own `DEEPSEEK_API_KEY` into every dsh pane before - * `applyEnvOverrides()` runs — so a non-granted owner who could set the base - * URL would have the operator's API key sent as a bearer credential to a host - * of their choosing. (`DEEPSEEK_API_KEY` itself stays overridable: supplying - * your OWN key removes privilege rather than granting it.) - * - `OMP_AUTH_BROKER_URL`/`OMP_AUTH_BROKER_TOKEN` are where omp resolves - * credentials from — the same shape as `DEEPSEEK_BASE_URL` above, reachable - * because `OMP_*` is an allowlisted prefix. Unlike DeepSeek, Codeman does not - * forward any operator-held key into an omp pane today (omp's provider - * credentials live in `~/.omp` config files, not env vars), so there is no - * known concrete exfiltration path yet — clamped defensively anyway, since a - * non-granted owner redirecting where a shared multi-tenant deployment - * resolves auth from is not something to allow silently (found in - * Ark0N/Codeman#353 review; omp's own knobs are otherwise mostly `PI_*`, - * already allowlisted for pi and not addressed here — see resolveOmpHome()). - */ -function ownerClampedEnvKeys(): string[] { - return enabledClis().flatMap((entry) => entry.capabilities.privilegedEnvKeys); -} - -/** - * Env-var half of the multi-user bypass clamp. - * - * `clampExternalCliBypassForOwner()` clamps the per-CLI CONFIG, and for every CLI - * but DeepSeek that is the whole story. Here it is not: `applyEnvOverrides()` runs - * AFTER `_configureCliEnv()` in tmux-manager, so an override sent on the SAME - * request lands last and wins, and a non-granted owner could restore - * `danger-full-access` on the very request the config clamp downgraded. - * - * Keys are DROPPED rather than rewritten: dropping falls through to what - * `_configureCliEnv()` exports, which is the clamped config and the server's own - * `DSH_HOME`, i.e. exactly the intended state. No-op in single-user mode and for a - * granted owner, like every other clamp here - * (`canUsernameRunPrivilegedCommands()` returns true when `!isMultiUserMode()`), - * and it returns the caller's own object untouched when there is nothing to strip. - */ -async function clampEnvOverridesForOwner( - owner: string | undefined, - envOverrides: Record<string, string> | undefined -): Promise<Record<string, string> | undefined> { - if (!envOverrides) return envOverrides; - const keys = ownerClampedEnvKeys(); - if (!keys.some((key) => key in envOverrides)) return envOverrides; - if (await canUsernameRunPrivilegedCommands(owner)) return envOverrides; - const clamped = { ...envOverrides }; - for (const key of keys) delete clamped[key]; - return clamped; -} - /** Test hook: the env-var half of the same multi-user safety gate. */ export const _clampEnvOverridesForOwner = clampEnvOverridesForOwner; diff --git a/src/web/schemas.ts b/src/web/schemas.ts index b5dabdd0..f0509f86 100644 --- a/src/web/schemas.ts +++ b/src/web/schemas.ts @@ -1161,6 +1161,20 @@ const NotificationEventSchema = z }) .optional(); +/** + * Body of `POST /api/reboot-restore/restore`. + * + * `sessionIds` restores a subset, and omitting it restores everything the caller + * can see. The ids are session ids from `GET /api/reboot-restore`, and an id the + * caller does not own is ignored rather than refused, matching how the session + * list scopes rather than 403s. + */ +export const RebootRestoreRequestSchema = z + .object({ + sessionIds: z.array(z.string().max(128)).max(200).optional(), + }) + .strict(); + export const SettingsUpdateSchema = z .object({ // User-facing product branding. This changes browser/UI copy only; package, diff --git a/src/web/server.ts b/src/web/server.ts index cf5927c0..065172d5 100644 --- a/src/web/server.ts +++ b/src/web/server.ts @@ -39,7 +39,9 @@ import { fileURLToPath } from 'node:url'; import { existsSync, mkdirSync, readFileSync, chmodSync, rmSync, statSync } from 'node:fs'; import fs from 'node:fs/promises'; import { execSync } from 'node:child_process'; -import { hostname as getHostname } from 'node:os'; +import { hostname as getHostname, uptime as osUptime } from 'node:os'; +import { looksLikeHostReboot, newestPersistedActivity, planRebootRestore } from '../reboot-restore.js'; +import { rebootRestoreRegistry } from './reboot-restore-registry.js'; import { dataPath, getDataDir, CODEMAN_INSTANCE } from '../config/instance.js'; import { normalizeBasePath, stripBasePath, joinBasePath } from '../config/base-path.js'; import { GLYPH, palette } from '../cli-style.js'; @@ -171,6 +173,7 @@ import { registerScheduledRoutes, registerHookEventRoutes, registerApprovalRoutes, + registerRebootRestoreRoutes, registerReadMyMindRoutes, registerStatusTelemetryRoutes, registerSystemRoutes, @@ -1060,6 +1063,7 @@ export class WebServer extends EventEmitter { registerScheduledRoutes(this.app, ctx); registerHookEventRoutes(this.app, ctx); registerApprovalRoutes(this.app, ctx); + registerRebootRestoreRoutes(this.app, ctx); registerReadMyMindRoutes(this.app, ctx); registerStatusTelemetryRoutes(this.app, ctx); registerSystemRoutes(this.app, ctx); @@ -2857,6 +2861,52 @@ export class WebServer extends EventEmitter { return false; } + /** + * Work out what a host reboot destroyed, and leave it on offer for the board. + * + * Runs inside `restoreMuxSessions()`, in the window after `reconcileSessions()` + * has reported the dead sessions and before `finalizeRestoredState()` prunes + * their records, so `state.json` is still the full picture here. That window is + * the only place the plan can be built, which is why the boot pass builds it + * even though nothing is rebuilt until a user clicks. + * + * Nothing is created here. The plan goes to `rebootRestoreRegistry`, the board + * offers it as a banner, and `web/routes/reboot-restore-routes` rebuilds what + * the user asks for. A wrong reboot guess therefore costs a line of text the + * user dismisses, not N CLI processes nobody asked for. + * + * @returns how many sessions are on offer. + */ + private planRebootRestoreOffer(dead: string[], livePaneCount: number): number { + if (dead.length === 0) return 0; + + const persisted = this.store.getSessions(); + if ( + !looksLikeHostReboot({ + livePaneCount, + deadSessionCount: dead.length, + uptimeSeconds: osUptime(), + newestPersistedActivityAt: newestPersistedActivity(persisted), + now: Date.now(), + }) + ) { + return 0; + } + + const { restore, skipped } = planRebootRestore(dead, persisted, (workingDir) => existsSync(workingDir)); + if (skipped.length > 0) { + console.log(`[Server] Reboot restore is passing over ${skipped.length} dead session(s):`); + for (const rejection of skipped) { + console.log(`[Server] ${rejection.sessionId}: ${rejection.reason}`); + } + } + rebootRestoreRegistry.set(restore); + if (restore.length > 0) { + console.log(`[Server] Host reboot detected; offering ${restore.length} session(s) for restore`); + } + return restore.length; + } + private async restoreMuxSessions(): Promise<boolean> { try { // Reconcile mux sessions to find which ones are still alive (also discovers unknown ones) @@ -2866,6 +2916,11 @@ export class WebServer extends EventEmitter { console.log(`[Server] Discovered ${discovered.length} unknown mux session(s)`); } + // Build the reboot-restore offer HERE: `dead` is only known after + // reconciliation, and the records it reads are pruned by + // `cleanupStaleSessions()` as soon as `finalizeRestoredState()` runs. + this.planRebootRestoreOffer(dead, alive.length); + if (alive.length > 0 || discovered.length > 0) { console.log(`[Server] Found ${alive.length + discovered.length} alive mux session(s) from previous run`); diff --git a/test/reboot-restore.test.ts b/test/reboot-restore.test.ts new file mode 100644 index 00000000..76506358 --- /dev/null +++ b/test/reboot-restore.test.ts @@ -0,0 +1,367 @@ +/** + * @fileoverview The decision half of reboot restore, and proof that the existing + * recovery construction path can CREATE a resumed pane. + * + * Three things are under test. `src/reboot-restore.ts` decides whether the + * machine rebooted and which dead sessions may be offered back. The plan + * registry in `src/web/reboot-restore-registry.ts` holds that offer between the + * boot that builds it and the click that spends it. The third is the claim the + * whole feature rests on: a `Session` built the way `restoreMuxSessions()` + * already builds one, but given no `muxSession` and a `resumeSessionId`, creates + * a fresh pane that resumes the old conversation. If that holds, the restore + * needs no new session-creation service. + * + * `reconcileSessions()` reports every session ALIVE under vitest, so the + * server's own boot pass cannot be reached from here. The decision logic is + * therefore driven directly, and the construction claim is driven through a real + * `Session` against the in-memory tmux layer vitest substitutes. + */ +import { mkdirSync, rmSync } from 'node:fs'; +import { homedir } from 'node:os'; +import { join } from 'node:path'; +import { afterEach, describe, expect, it } from 'vitest'; + +import { Session } from '../src/session.js'; +import { TmuxManager } from '../src/tmux-manager.js'; +import type { SessionState } from '../src/types.js'; +import { + looksLikeHostReboot, + newestPersistedActivity, + planRebootRestore, + rejectAlreadyLive, + resolveResumeConversationId, + type RebootRestoreEntry, +} from '../src/reboot-restore.js'; +import { RebootRestoreRegistry } from '../src/web/reboot-restore-registry.js'; + +const HOUR = 60 * 60 * 1000; +const NOW = 1_760_000_000_000; + +function persistedSession(overrides: Partial<SessionState> & { id: string }): SessionState { + return { + pid: 99999, + status: 'idle', + workingDir: '/tmp/spike', + currentTaskId: null, + createdAt: NOW - 4 * HOUR, + lastActivityAt: NOW - 2 * HOUR, + mode: 'claude', + ...overrides, + } as SessionState; +} + +describe('reboot detection', () => { + const base = { + livePaneCount: 0, + deadSessionCount: 2, + // The host came up 10 minutes ago, well after the sessions were last active. + uptimeSeconds: 600, + newestPersistedActivityAt: NOW - 2 * HOUR, + now: NOW, + }; + + it('calls it a reboot when the socket is empty and the host booted after the last activity', () => { + expect(looksLikeHostReboot(base)).toBe(true); + }); + + it('refuses when some panes survived, which is an ordinary server restart', () => { + expect(looksLikeHostReboot({ ...base, livePaneCount: 3 })).toBe(false); + }); + + it('refuses on a long-uptime host, where someone wiped the tmux socket by hand', () => { + // Up for 30 days: the sessions were active long AFTER this boot, so the panes + // went away for some reason other than the machine restarting. + expect(looksLikeHostReboot({ ...base, uptimeSeconds: 30 * 24 * 60 * 60 })).toBe(false); + }); + + it('refuses when nothing died', () => { + expect(looksLikeHostReboot({ ...base, deadSessionCount: 0 })).toBe(false); + }); + + it('reads the newest activity stamp across the persisted records', () => { + const persisted = { + a: persistedSession({ id: 'a', lastActivityAt: NOW - 5 * HOUR }), + b: persistedSession({ id: 'b', lastActivityAt: NOW - 1 * HOUR }), + }; + expect(newestPersistedActivity(persisted)).toBe(NOW - 1 * HOUR); + }); +}); + +describe('which dead sessions may be rebuilt', () => { + it('rebuilds a session that was simply running when the power went out', () => { + const persisted = { live: persistedSession({ id: 'live', status: 'busy' }) }; + const plan = planRebootRestore(['live'], persisted, () => true); + expect(plan.restore.map((s) => s.sessionId)).toEqual(['live']); + }); + + it('never revives a session the user killed while pinned (COD-142 demotes it to stopped)', () => { + const persisted = { killed: persistedSession({ id: 'killed', status: 'stopped', pinned: true }) }; + const plan = planRebootRestore(['killed'], persisted, () => true); + expect(plan.restore).toEqual([]); + expect(plan.skipped).toEqual([{ sessionId: 'killed', reason: 'intentionally-ended' }]); + }); + + it('never revives a session whose record an unpinned kill already deleted', () => { + const plan = planRebootRestore(['gone'], {}, () => true); + expect(plan.restore).toEqual([]); + expect(plan.skipped).toEqual([{ sessionId: 'gone', reason: 'no-persisted-record' }]); + }); + + it('never revives a pane whose PTY-exit breaker had tripped', () => { + const persisted = { crashy: persistedSession({ id: 'crashy', respawnBlocked: true }) }; + expect(planRebootRestore(['crashy'], persisted, () => true).skipped[0].reason).toBe('respawn-blocked'); + }); + + it('leaves remote sessions to the COD-108 reconnect watcher', () => { + const persisted = { + r: persistedSession({ + id: 'r', + remote: { hostId: 'h', host: 'example.test', username: 'u', sessionName: 'n', owned: true }, + } as Partial<SessionState> & { id: string }), + }; + expect(planRebootRestore(['r'], persisted, () => true).skipped[0].reason).toBe('remote-or-docker'); + }); + + it('leaves docker sessions alone, since the container may not be up', () => { + const persisted = { + d: persistedSession({ id: 'd', docker: { containerId: 'abc', caseId: 'c' } } as Partial<SessionState> & { + id: string; + }), + }; + expect(planRebootRestore(['d'], persisted, () => true).skipped[0].reason).toBe('remote-or-docker'); + }); + + it('skips a CLI whose history the claude transcript reader does not understand', () => { + const persisted = { c: persistedSession({ id: 'c', mode: 'codex' }) }; + expect(planRebootRestore(['c'], persisted, () => true).skipped[0].reason).toBe('unsupported-mode'); + }); +}); + +describe('a workspace that is no longer on disk', () => { + it('is kept out of the offer, so a click cannot scaffold a deleted repo', () => { + const persisted = { gone: persistedSession({ id: 'gone', workingDir: '/tmp/deleted-repo' }) }; + const plan = planRebootRestore(['gone'], persisted, () => false); + expect(plan.restore).toEqual([]); + expect(plan.skipped).toEqual([{ sessionId: 'gone', reason: 'workspace-missing' }]); + }); + + it('is judged per session, not for the batch', () => { + const persisted = { + kept: persistedSession({ id: 'kept', workingDir: '/tmp/still-here' }), + gone: persistedSession({ id: 'gone', workingDir: '/tmp/deleted-repo' }), + }; + const plan = planRebootRestore(['kept', 'gone'], persisted, (dir) => dir === '/tmp/still-here'); + expect(plan.restore.map((entry) => entry.sessionId)).toEqual(['kept']); + expect(plan.skipped.map((s) => s.reason)).toEqual(['workspace-missing']); + }); +}); + +describe('a conversation that came back on its own before the click', () => { + const entry: RebootRestoreEntry = { + sessionId: 'abc', + workingDir: '/tmp/spike', + mode: 'claude', + resumeConversationId: 'conv-1', + state: persistedSession({ id: 'abc' }), + }; + + it('is skipped when the user resumed it by hand from the Resume list', () => { + // Same conversation, different session id: the Resume list creates a NEW id. + const result = rejectAlreadyLive([entry], new Set(['other']), new Set(['conv-1'])); + expect(result.restore).toEqual([]); + expect(result.skipped).toEqual([{ sessionId: 'abc', reason: 'already-live' }]); + }); + + it('is skipped when a session with that id is already on the board', () => { + const result = rejectAlreadyLive([entry], new Set(['abc']), new Set()); + expect(result.skipped).toEqual([{ sessionId: 'abc', reason: 'already-live' }]); + }); + + it('is rebuilt when neither its id nor its conversation is live', () => { + const result = rejectAlreadyLive([entry], new Set(['other']), new Set(['conv-other'])); + expect(result.restore.map((e) => e.sessionId)).toEqual(['abc']); + expect(result.skipped).toEqual([]); + }); +}); + +describe('the plan the banner spends', () => { + const all = () => true; + const entryFor = (sessionId: string, owner?: string): RebootRestoreEntry => ({ + sessionId, + owner, + workingDir: '/tmp/spike', + mode: 'claude', + resumeConversationId: `conv-${sessionId}`, + state: persistedSession({ id: sessionId, owner }), + }); + + it('hands an entry to the first caller and nothing to the second', () => { + const registry = new RebootRestoreRegistry(); + registry.set([entryFor('a'), entryFor('b')]); + expect(registry.take(all).map((e) => e.sessionId)).toEqual(['a', 'b']); + // The double-click: two panes on one conversation is what this prevents. + expect(registry.take(all)).toEqual([]); + }); + + it('spends only the ids a caller asked for', () => { + const registry = new RebootRestoreRegistry(); + registry.set([entryFor('a'), entryFor('b')]); + expect(registry.take(all, ['b']).map((e) => e.sessionId)).toEqual(['b']); + expect(registry.list(all).map((e) => e.sessionId)).toEqual(['a']); + }); + + it("shows a user their own sessions and leaves another owner's alone", () => { + const registry = new RebootRestoreRegistry(); + registry.set([entryFor('mine', 'alice'), entryFor('theirs', 'bob')]); + const asAlice = (owner: string | undefined) => owner === 'alice'; + expect(registry.list(asAlice).map((e) => e.sessionId)).toEqual(['mine']); + expect(registry.take(asAlice).map((e) => e.sessionId)).toEqual(['mine']); + // Bob's entry is still on offer for Bob. + expect(registry.list(() => true).map((e) => e.sessionId)).toEqual(['theirs']); + }); + + it('puts back an entry that no pane was created for', () => { + const registry = new RebootRestoreRegistry(); + registry.set([entryFor('a')]); + const taken = registry.take(all); + registry.restore(taken); + expect(registry.list(all).map((e) => e.sessionId)).toEqual(['a']); + }); + + it('runs one restore at a time', () => { + const registry = new RebootRestoreRegistry(); + expect(registry.beginSpending()).toBe(true); + expect(registry.beginSpending()).toBe(false); + registry.endSpending(); + expect(registry.beginSpending()).toBe(true); + }); + + it('drops what a dismiss cleared', () => { + const registry = new RebootRestoreRegistry(); + registry.set([entryFor('a'), entryFor('b')]); + expect(registry.clear(all)).toBe(2); + expect(registry.list(all)).toEqual([]); + }); + + it('forgets a plan nobody took for a day', () => { + const registry = new RebootRestoreRegistry(); + registry.set([entryFor('a')]); + const dayLater = Date.now() + 25 * HOUR; + const realNow = Date.now; + Date.now = () => dayLater; + try { + expect(registry.list(all)).toEqual([]); + } finally { + Date.now = realNow; + } + }); +}); + +describe('which conversation a rebuilt pane resumes', () => { + it('prefers the chain tail, the conversation the CLI reported last', () => { + const state = persistedSession({ + id: 'sess-1', + resumeSessionId: 'launch-id', + claudeSessionChain: ['launch-id', 'after-clear'], + }); + expect(resolveResumeConversationId(state)).toBe('after-clear'); + }); + + it('falls back to the id the session originally resumed', () => { + const state = persistedSession({ id: 'sess-1', resumeSessionId: 'resumed-id' }); + expect(resolveResumeConversationId(state)).toBe('resumed-id'); + }); + + it('falls back to the session id, which is what Claude was launched with', () => { + expect(resolveResumeConversationId(persistedSession({ id: 'sess-1' }))).toBe('sess-1'); + }); +}); + +describe('the recovery construction path can create a resumed pane', () => { + const workingDir = join(homedir(), 'codeman-cases', 'reboot-restore-spike'); + const sessions: Session[] = []; + + afterEach(() => { + for (const s of sessions.splice(0)) s.stop(); + rmSync(workingDir, { recursive: true, force: true }); + }); + + /** Built exactly as the reboot pass builds one: no `muxSession`, plus a resume id. */ + function rebuildFromPersistedState(state: SessionState, mux: TmuxManager): Session { + mkdirSync(workingDir, { recursive: true }); + const session = new Session({ + id: state.id, + workingDir, + mode: state.mode, + name: state.name, + createdAt: state.createdAt, + mux, + useMux: true, + resumeSessionId: resolveResumeConversationId(state), + owner: state.owner, + lastActivityAt: state.lastActivityAt, + claudeSessionChain: state.claudeSessionChain, + }); + sessions.push(session); + return session; + } + + it('creates a NEW mux session rather than needing one to attach to', async () => { + const mux = new TmuxManager(); + const state = persistedSession({ id: 'aaaaaaa1-1111-4111-8111-111111111111', name: 'w1-spike' }); + const session = rebuildFromPersistedState(state, mux); + + expect(mux.getSessions()).toHaveLength(0); + await session.startInteractive(); + + const created = mux.getSessions(); + expect(created).toHaveLength(1); + expect(created[0].sessionId).toBe('aaaaaaa1-1111-4111-8111-111111111111'); + expect(created[0].workingDir).toBe(workingDir); + }); + + it('comes back pointed at the conversation the pane was holding', async () => { + const mux = new TmuxManager(); + const state = persistedSession({ + id: 'aaaaaaa2-2222-4222-8222-222222222222', + resumeSessionId: 'launch-id', + claudeSessionChain: ['launch-id', 'after-clear'], + }); + const session = rebuildFromPersistedState(state, mux); + + await session.startInteractive(); + + // The chain tail wins: a `/clear` before the reboot moved the CLI off the launch id. + expect(session.claudeSessionId).toBe('after-clear'); + }); + + it('comes back idle, with no prompt sent and no autonomous loop armed', async () => { + const mux = new TmuxManager(); + const state = persistedSession({ + id: 'aaaaaaa3-3333-4333-8333-333333333333', + ralphEnabled: true, + respawnEnabled: true, + }); + const session = rebuildFromPersistedState(state, mux); + + await session.startInteractive(); + + // No prompt was queued: nothing is waiting on a task. The status itself is not + // assertable here, because the test PTY echoes and the activity detector reads + // that echo as work; in production the pane settles once the CLI finishes booting. + expect(session.currentTaskId).toBeNull(); + // The pass never touches the tracker, so a persisted Ralph loop stays cold. + expect(session.ralphTracker.enabled).toBe(false); + }); + + it('keeps the owner it was persisted with, there being no request to read one from', async () => { + const mux = new TmuxManager(); + const state = persistedSession({ id: 'aaaaaaa4-4444-4444-8444-444444444444', owner: 'alice' }); + const session = rebuildFromPersistedState(state, mux); + + await session.startInteractive(); + + expect(session.owner).toBe('alice'); + expect(mux.getSessions()[0].owner).toBe('alice'); + }); +}); diff --git a/test/routes/reboot-restore-routes.test.ts b/test/routes/reboot-restore-routes.test.ts new file mode 100644 index 00000000..c16cddac --- /dev/null +++ b/test/routes/reboot-restore-routes.test.ts @@ -0,0 +1,175 @@ +/** + * Reboot-restore route tests (src/web/routes/reboot-restore-routes.ts) via + * app.inject(), no live port. + * + * Every entry these tests put on offer names a workspace that does not exist, so + * the route's click-time workspace check rejects it before any `Session` is + * constructed. That keeps the tests on the route's own guards — taking, scoping, + * single-flighting and re-checking — and leaves pane creation to + * test/reboot-restore.test.ts, which drives a real `Session` for it. + * + * The routes read the process-wide `rebootRestoreRegistry` singleton, so every + * test resets it; a leaked entry would bleed into the next one. + */ +import { describe, it, expect, afterEach } from 'vitest'; +import Fastify, { type FastifyInstance } from 'fastify'; +import fastifyCookie from '@fastify/cookie'; +import { registerRebootRestoreRoutes } from '../../src/web/routes/reboot-restore-routes.js'; +import { rebootRestoreRegistry } from '../../src/web/reboot-restore-registry.js'; +import { installRouteErrorHandler } from '../../src/web/route-error-handler.js'; +import { httpStatusForErrorCode, type ApiErrorCode } from '../../src/types.js'; +import { createMockRouteContext } from '../mocks/index.js'; +import type { RebootRestoreEntry } from '../../src/reboot-restore.js'; +import type { SessionState } from '../../src/types.js'; + +async function createHarness(authUser?: { username: string; role: 'admin' | 'user' }): Promise<FastifyInstance> { + const app = Fastify({ logger: false }); + await app.register(fastifyCookie); + if (authUser) { + app.addHook('onRequest', async (req) => { + (req as unknown as { authUser: typeof authUser }).authUser = authUser; + }); + } + registerRebootRestoreRoutes(app, createMockRouteContext() as never); + + app.addHook('preSerialization', (req, reply, payload: unknown, done) => { + if (!req.url.startsWith('/api')) return done(null, payload); + if (payload === null || typeof payload !== 'object') return done(null, payload); + const p = payload as { success?: unknown; errorCode?: unknown }; + if (p.success === false) { + if (reply.statusCode === 200 && typeof p.errorCode === 'string') { + reply.code(httpStatusForErrorCode(p.errorCode as ApiErrorCode)); + } + return done(null, payload); + } + if (p.success === true) return done(null, payload); + return done(null, { success: true, data: payload }); + }); + + installRouteErrorHandler(app); + await app.ready(); + return app; +} + +/** An entry whose workspace is deliberately absent, so no pane is ever created. */ +function offerEntry(sessionId: string, owner?: string): RebootRestoreEntry { + return { + sessionId, + name: `session ${sessionId}`, + workingDir: `/tmp/codeman-reboot-restore-missing/${sessionId}`, + owner, + mode: 'claude', + resumeConversationId: `conv-${sessionId}`, + state: { + id: sessionId, + pid: null, + status: 'idle', + workingDir: `/tmp/codeman-reboot-restore-missing/${sessionId}`, + currentTaskId: null, + createdAt: 1_760_000_000_000, + mode: 'claude', + owner, + } as SessionState, + }; +} + +afterEach(() => { + rebootRestoreRegistry.reset(); +}); + +describe('GET /api/reboot-restore', () => { + it('reports nothing when no reboot left anything behind', async () => { + const app = await createHarness(); + const res = await app.inject({ method: 'GET', url: '/api/reboot-restore' }); + expect(res.statusCode).toBe(200); + expect(res.json().data.sessions).toEqual([]); + await app.close(); + }); + + it('names what is on offer, and says the scrollback is not coming back', async () => { + rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]); + const app = await createHarness(); + const body = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data; + expect(body.sessions.map((s: { id: string }) => s.id)).toEqual(['a', 'b']); + expect(body.scrollbackRestored).toBe(false); + await app.close(); + }); + + it('never carries the persisted record itself to the browser', async () => { + rebootRestoreRegistry.set([offerEntry('a', 'alice')]); + const app = await createHarness({ username: 'alice', role: 'admin' }); + const body = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data; + expect(Object.keys(body.sessions[0]).sort()).toEqual(['id', 'mode', 'name', 'owner', 'workingDir']); + expect(body.sessions[0].state).toBeUndefined(); + await app.close(); + }); +}); + +describe('POST /api/reboot-restore/restore', () => { + it('spends the offer, so a second click finds nothing left to spend', async () => { + rebootRestoreRegistry.set([offerEntry('a')]); + const app = await createHarness(); + + const first = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data; + // The workspace is gone, so nothing was rebuilt — but the entry was taken. + expect(first.restored).toEqual([]); + expect(first.skipped).toEqual([{ sessionId: 'a', reason: 'workspace-missing' }]); + + const second = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data; + expect(second.restored).toEqual([]); + expect(second.skipped).toEqual([]); + await app.close(); + }); + + it('spends only the sessions the click named', async () => { + rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]); + const app = await createHarness(); + + const res = await app.inject({ + method: 'POST', + url: '/api/reboot-restore/restore', + payload: { sessionIds: ['b'] }, + }); + expect(res.json().data.skipped).toEqual([{ sessionId: 'b', reason: 'workspace-missing' }]); + + const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data; + expect(left.sessions.map((s: { id: string }) => s.id)).toEqual(['a']); + await app.close(); + }); + + it('refuses a body it does not recognise rather than guessing', async () => { + const app = await createHarness(); + const res = await app.inject({ + method: 'POST', + url: '/api/reboot-restore/restore', + payload: { sessionIds: 'not-an-array' }, + }); + expect(res.statusCode).toBeGreaterThanOrEqual(400); + await app.close(); + }); + + it('turns a second concurrent restore away rather than interleaving it', async () => { + rebootRestoreRegistry.set([offerEntry('a')]); + // Claimed by a restore already in flight. + expect(rebootRestoreRegistry.beginSpending()).toBe(true); + const app = await createHarness(); + const res = await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} }); + expect(res.statusCode).toBe(409); + rebootRestoreRegistry.endSpending(); + await app.close(); + }); +}); + +describe('POST /api/reboot-restore/dismiss', () => { + it('drops the offer and leaves the banner with nothing to show', async () => { + rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]); + const app = await createHarness(); + + const res = await app.inject({ method: 'POST', url: '/api/reboot-restore/dismiss', payload: {} }); + expect(res.json().data.dismissed).toBe(2); + + const after = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data; + expect(after.sessions).toEqual([]); + await app.close(); + }); +}); From fbede5cd2a20bfc50074c113da7247adb09adf0c Mon Sep 17 00:00:00 2001 From: Michael Grundberg <michael.grundberg@irisity.com> Date: Wed, 16 Sep 2026 11:44:37 +0200 Subject: [PATCH 11/28] fix(sessions): act on the dual review of the reboot-restore route Fifteen findings from two independent reviews of #442, three of them blocking. Every one is addressed here. The three blockers all sat in the restore route. A rebuild that threw after addSession left a registered session with no pane behind it, visible on the board, holding a layout slot and written to state.json, with its plan entry already spent; the catch now cleans the session up and puts the entry back. The loop checked neither the global nor the per-user session cap, so one click could take a board past a documented limit; capacity is now re-checked per iteration, because the loop is itself creating the sessions it counts. Worst of the three, a rebuilt session carried none of the state its constructor has no parameter for and then persisted itself over the record that held it, zeroing token and cost totals and dropping the pin. The pin matters most: pruning keeps a record only while it is pinned, so discarding it handed the record to the next stale sweep. A new reapplyPersistedSessionState() on the session port restores the pin, the token totals, auto-compact, auto-clear, auto-resume, nice priority, the flicker filter and the custom-model selection, and it runs before both startInteractive and the first persist. The rest, in the order they bite a user. Every rebuild failure was reported as workspace-missing, so the banner told users their repo was gone when the agent had simply failed to start; there are now distinct reasons, and the toast names each one. The client read restored and skipped off the outer response object rather than through the uniform envelope, so every count came back zero and neither toast ever fired. A board left open across the reboot never learned an offer existed, because the banner was seeded only on the page-load path; it now re-reads on every SSE init. The workspace check was existence-only, skipping the multi-user confinement that the create route applies, so a withdrawn grant would not be noticed. The banner had no phone breakpoint while its text was nowrap and its buttons could not shrink. Smaller: a missing workspace is now re-offered rather than dropped, while an already-open conversation is dropped rather than re-offered forever; a throw anywhere in the route returns the unspent entries instead of discarding the plan; the single flight is keyed by owner, since take() already stops two callers receiving one entry; the env clamp's header no longer claims a protection it cannot provide on this path today, and names the check that does bite; the three endpoints are documented in docs/api-reference.md; and the module header now says that os.uptime() reads the host's clock, so the feature is effectively off inside a container. The review also explained why the tests missed all of this: they proved the construction claim through their own copy of the construction rather than through the route, and the route tests used workspaces that did not exist, so no Session was ever built. test/routes/reboot-restore-rebuild-failure.ts mocks the Session module to drive the route's real path, and covers the cleanup, the reason reported, the re-application ordering, the broadcast and the caps. The mock route context gains the port method and the mux call the route needs. Refs #411 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- docs/api-reference.md | 201 ++++++++++------- src/reboot-restore.ts | 20 +- src/session-env-clamp.ts | 14 +- src/web/ports/session-port.ts | 12 + src/web/public/app.js | 10 +- src/web/public/mobile.css | 35 +++ src/web/public/reboot-restore-ui.js | 43 +++- src/web/reboot-restore-registry.ts | 39 ++-- src/web/routes/reboot-restore-routes.ts | 72 +++++- src/web/server.ts | 48 ++++ test/mocks/mock-route-context.ts | 2 + .../reboot-restore-rebuild-failure.test.ts | 210 ++++++++++++++++++ test/routes/reboot-restore-routes.test.ts | 47 +++- 13 files changed, 632 insertions(+), 121 deletions(-) create mode 100644 test/routes/reboot-restore-rebuild-failure.test.ts diff --git a/docs/api-reference.md b/docs/api-reference.md index dcd0d631..85349ee3 100644 --- a/docs/api-reference.md +++ b/docs/api-reference.md @@ -66,17 +66,17 @@ The single source of truth is `ErrorStatus` / `httpStatusForErrorCode()` in `src/types/api.ts`. Clients should branch on `errorCode` (stable) and may rely on the HTTP status. -| `errorCode` | HTTP | Meaning | -|-------------|------|---------| -| `INVALID_INPUT` | 400 | Malformed request / failed validation | -| `UNAUTHORIZED` | 401 | Authentication required or failed | -| `NOT_FOUND` | 404 | Resource does not exist | -| `SESSION_BUSY` | 409 | Session is busy | -| `CONFLICT` | 409 | Conflicts with current state (e.g. already running) | -| `ALREADY_EXISTS` | 409 | Resource already exists | -| `OPERATION_FAILED` | 422 | Well-formed but could not be completed | -| `RATE_LIMITED` | 429 | Too many requests | -| `INTERNAL_ERROR` | 500 | Unexpected server error | +| `errorCode` | HTTP | Meaning | +| ------------------ | ---- | --------------------------------------------------- | +| `INVALID_INPUT` | 400 | Malformed request / failed validation | +| `UNAUTHORIZED` | 401 | Authentication required or failed | +| `NOT_FOUND` | 404 | Resource does not exist | +| `SESSION_BUSY` | 409 | Session is busy | +| `CONFLICT` | 409 | Conflicts with current state (e.g. already running) | +| `ALREADY_EXISTS` | 409 | Resource already exists | +| `OPERATION_FAILED` | 422 | Well-formed but could not be completed | +| `RATE_LIMITED` | 429 | Too many requests | +| `INTERNAL_ERROR` | 500 | Unexpected server error | Adding a new error code is non-breaking; removing or renaming one is a major change. @@ -87,10 +87,10 @@ exist because SSE is Codeman's only other "tell me when" channel, and an agent driving the API from a shell tool cannot practically hold a stream and parse events inline. -| Call | Blocks until | -|------|--------------| -| `GET /api/v1/sessions/:id/wait` | one of a set of lifecycle signals fires | -| `GET /api/v1/sessions/:id/wait-output` | a literal string appears in the session's output | +| Call | Blocks until | +| --------------------------------------------- | -------------------------------------------------- | +| `GET /api/v1/sessions/:id/wait` | one of a set of lifecycle signals fires | +| `GET /api/v1/sessions/:id/wait-output` | a literal string appears in the session's output | | `POST /api/v1/sessions/:id/input` with `wait` | the input is delivered **and then** a signal fires | `POST .../input` with `wait` is not the same as a `POST` followed by a separate @@ -140,13 +140,13 @@ contract is a **marker unique to each call** (`MARK="DONE_$RANDOM"`, send ### Signals -| Signal | Source | Actually fires for | -|--------|--------|--------------------| -| `idle` | the session's own `idle` event | `claude`: yes, on ❯-prompt detection after activity. `shell`: **once only**, ~500 ms after start, and never again. External CLIs: not guaranteed (they render their own TUIs and readiness is output stabilization) | -| `working` | the session's own `working` event | `claude` only in practice (spinner and work-keyword detection are Claude output formats) | -| `stop` | the Claude Code `stop` hook, the definitive end-of-turn signal | `claude` only | -| `blocked` | a `permission_prompt` or `elicitation_dialog` hook | `claude` only, and rarer than it looks: see below | -| `exit` | no process is behind the session | every mode | +| Signal | Source | Actually fires for | +| --------- | -------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `idle` | the session's own `idle` event | `claude`: yes, on ❯-prompt detection after activity. `shell`: **once only**, ~500 ms after start, and never again. External CLIs: not guaranteed (they render their own TUIs and readiness is output stabilization) | +| `working` | the session's own `working` event | `claude` only in practice (spinner and work-keyword detection are Claude output formats) | +| `stop` | the Claude Code `stop` hook, the definitive end-of-turn signal | `claude` only | +| `blocked` | a `permission_prompt` or `elicitation_dialog` hook | `claude` only, and rarer than it looks: see below | +| `exit` | no process is behind the session | every mode | `stop` is the signal to orchestrate on where it exists; `idle` is a heuristic fallback that can flap mid-turn when a spinner pauses. The default set when `until` @@ -156,12 +156,12 @@ can no longer happen). On a `claude` worker, prefer an explicit `until=stop,exit once the session is up: the default set's `idle` also resolves on a spinner pause, and on a fresh session the **startup** `idle` (emitted when the CLI first comes up) can land inside your first wait window and report a turn that never ran. Measured: -a session parked on the trust dialog emits no *further* `idle`, so it is the +a session parked on the trust dialog emits no _further_ `idle`, so it is the startup transition, not the dialog, that produces the false success below. ⚠️ **`exit` means "nothing is running", which includes "not started yet".** The server answers from `pid === null` plus a mux-layer pane-death probe, and that -covers a session that exited — including a worker that died *inside* its tmux pane +covers a session that exited — including a worker that died _inside_ its tmux pane while the local attach client (and therefore `pid`) lives on — one that was detached, and one that was **created but never started**. So the first wait after `POST /api/v1/sessions` returns `{"signal":"exit","immediate":true}` in @@ -184,7 +184,7 @@ blocked, and polling `blocked` alone will sit at its timeout. ⚠️ **On a `shell` session, only `exit` and marker-matching are dependable.** A shell session emits its one `idle` at startup and then stays `status: "idle"` forever, -whatever the pane is doing, so it never emits a *transition*. Since send-and-wait +whatever the pane is doing, so it never emits a _transition_. Since send-and-wait requires a transition (and so does `fresh=1`), both can only time out there: a documented default `wait` on a shell worker running `sleep 4` times out at the full 25 s. Synchronize hook-less sessions with `wait-output` and a unique marker @@ -218,11 +218,11 @@ with `from=buffer` keeps matching long after the dialog is gone. A worked versio ### `GET /api/v1/sessions/:id/wait` -| Param | Type | Default | Notes | -|-------|------|---------|-------| -| `until` | comma-separated list of `idle,working,stop,blocked,exit` | `stop,idle,exit` | resolves on the first to fire. An unknown token is a `400` naming it, never a silent fallback | -| `timeout` | positive integer ms | `60000` | **validated first, clamped second.** `0`, a negative value and a fractional value are all `400`s, not clamps; a valid value outside `[1000, 600000]` is clamped and echoed as `wait.timeoutMs` | -| `fresh` | `0` \| `1` \| `false` \| `true` | `0` | `1` requires an actual transition, ignoring the state at call time | +| Param | Type | Default | Notes | +| --------- | -------------------------------------------------------- | ---------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `until` | comma-separated list of `idle,working,stop,blocked,exit` | `stop,idle,exit` | resolves on the first to fire. An unknown token is a `400` naming it, never a silent fallback | +| `timeout` | positive integer ms | `60000` | **validated first, clamped second.** `0`, a negative value and a fractional value are all `400`s, not clamps; a valid value outside `[1000, 600000]` is clamped and echoed as `wait.timeoutMs` | +| `fresh` | `0` \| `1` \| `false` \| `true` | `0` | `1` requires an actual transition, ignoring the state at call time | ```bash curl -s "$API/api/v1/sessions/$SID/wait?until=stop,exit&timeout=60000" @@ -239,12 +239,12 @@ a plain signal wait, so check the endpoint path before blaming the parameters. ### `GET /api/v1/sessions/:id/wait-output` -| Param | Type | Default | Notes | -|-------|------|---------|-------| -| `match` | literal string, 1 to 200 chars | required | substring match against the PTY stream with ANSI escapes stripped. A match spanning two PTY chunks is found | -| `nocase` | `0` \| `1` \| `false` \| `true` | `0` | case-insensitive compare. The returned snippet keeps the terminal's original casing | -| `from` | `now` \| `buffer` | `now` | `buffer` scans the tail of the existing terminal buffer (bounded, 256 KB by default) before blocking | -| `timeout` | positive integer ms | `60000` | same validation and clamp as `/wait` | +| Param | Type | Default | Notes | +| --------- | ------------------------------- | -------- | ----------------------------------------------------------------------------------------------------------- | +| `match` | literal string, 1 to 200 chars | required | substring match against the PTY stream with ANSI escapes stripped. A match spanning two PTY chunks is found | +| `nocase` | `0` \| `1` \| `false` \| `true` | `0` | case-insensitive compare. The returned snippet keeps the terminal's original casing | +| `from` | `now` \| `buffer` | `now` | `buffer` scans the tail of the existing terminal buffer (bounded, 256 KB by default) before blocking | +| `timeout` | positive integer ms | `60000` | same validation and clamp as `/wait` | **Matching is literal, never a pattern.** A `regex` parameter is rejected with a `400` rather than ignored, so a caller that assumed otherwise finds out immediately @@ -296,10 +296,10 @@ hand-written query string decodes to a space. Two optional fields on the existing endpoint: -| Field | Type | Notes | -|-------|------|-------| -| `wait` | `true` or the same comma grammar as `until` | `true` means the default signal set. Omitted keeps the historical fire-and-forget behavior, unchanged. `null`, `false` and an empty string are all read as **absent**, not as an error and not as "wait for the default" | -| `waitTimeout` | positive integer ms | same validation **and** clamp as `timeout`: `0`, a negative and a fractional value are `400`s, anything valid is clamped into `[1000, 600000]` and echoed as `wait.timeoutMs` | +| Field | Type | Notes | +| ------------- | ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `wait` | `true` or the same comma grammar as `until` | `true` means the default signal set. Omitted keeps the historical fire-and-forget behavior, unchanged. `null`, `false` and an empty string are all read as **absent**, not as an error and not as "wait for the default" | +| `waitTimeout` | positive integer ms | same validation **and** clamp as `timeout`: `0`, a negative and a fractional value are `400`s, anything valid is clamped into `[1000, 600000]` and echoed as `wait.timeoutMs` | Both are `nullish`, so an explicit `null` from `JSON.stringify` is accepted as "absent" rather than failing validation. That is deliberate: `.optional()` would @@ -330,16 +330,24 @@ All three nest the wait result under `data.wait`, so one client helper works aga any of them: ```json -{ "success": true, "data": { - "sessionId": "28325fd3-caa7-4178-82bf-87dfebf0f464", - "status": "idle", - "limitPaused": false, - "wait": { - "signal": "stop", "until": ["stop", "idle", "exit"], - "timedOut": false, "immediate": false, "ended": false, "aborted": false, - "waitedMs": 8421, "timeoutMs": 60000 +{ + "success": true, + "data": { + "sessionId": "28325fd3-caa7-4178-82bf-87dfebf0f464", + "status": "idle", + "limitPaused": false, + "wait": { + "signal": "stop", + "until": ["stop", "idle", "exit"], + "timedOut": false, + "immediate": false, + "ended": false, + "aborted": false, + "waitedMs": 8421, + "timeoutMs": 60000 + } } -}} +} ``` `POST .../input` returns the same `wait` object alongside `delivered`, `duplicate`, @@ -353,21 +361,21 @@ redelivery (harmless, the turn it refers to may be long over), while with client that reads `delivered === false` as "duplicate" silently treats a failed send as a success. -| Field | Type | Meaning | -|-------|------|---------| -| `wait.signal` | signal \| `null` | the signal that fired (`/wait` and `/input` only) | -| `wait.until` | array of signals | what the server actually waited on, after narrowing the default set for the session's mode (`/wait` and `/input` only) | -| `wait.matched` | boolean | the string appeared (`/wait-output` only) | -| `wait.match` | string | the literal that was searched for (`/wait-output` only) | -| `wait.snippet` | string \| `null` | bounded window of output around the match, blank runs collapsed for readability (`/wait-output` only) | -| `wait.timedOut` | boolean | the wait hit its timeout. Still a `200` | -| `wait.immediate` | boolean | the condition already held at call time, so nothing was waited for (`waitedMs` is 0) | -| `wait.ended` | boolean | the session went away (deleted or torn down) before the condition was met | -| `wait.aborted` | boolean | the client hung up, so the waiter was released without resolving — and by that definition a client never reads `true`. When the **server** abandons a wait itself (send-and-wait against a session with no PTY), it answers in about a millisecond with `ended: true`, `delivered: false`, `duplicate: false` and `aborted: false`: `delivered`/`ended` carry that story, and `aborted` stays the transport flag. Present for completeness; treat a `true` as "this wait answered nothing", never as an outcome | -| `wait.waitedMs` | number | wall-clock ms actually spent waiting | -| `wait.timeoutMs` | number | the timeout **after clamping**, which is what was applied | -| `status` | `SessionStatus` | the session's status after the wait, so a caller that timed out still learns where things stand | -| `limitPaused` | boolean | the session is paused on a usage limit and will emit nothing until its reset, so a timeout here is expected rather than a stall worth retrying hard | +| Field | Type | Meaning | +| ---------------- | ---------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `wait.signal` | signal \| `null` | the signal that fired (`/wait` and `/input` only) | +| `wait.until` | array of signals | what the server actually waited on, after narrowing the default set for the session's mode (`/wait` and `/input` only) | +| `wait.matched` | boolean | the string appeared (`/wait-output` only) | +| `wait.match` | string | the literal that was searched for (`/wait-output` only) | +| `wait.snippet` | string \| `null` | bounded window of output around the match, blank runs collapsed for readability (`/wait-output` only) | +| `wait.timedOut` | boolean | the wait hit its timeout. Still a `200` | +| `wait.immediate` | boolean | the condition already held at call time, so nothing was waited for (`waitedMs` is 0) | +| `wait.ended` | boolean | the session went away (deleted or torn down) before the condition was met | +| `wait.aborted` | boolean | the client hung up, so the waiter was released without resolving — and by that definition a client never reads `true`. When the **server** abandons a wait itself (send-and-wait against a session with no PTY), it answers in about a millisecond with `ended: true`, `delivered: false`, `duplicate: false` and `aborted: false`: `delivered`/`ended` carry that story, and `aborted` stays the transport flag. Present for completeness; treat a `true` as "this wait answered nothing", never as an outcome | +| `wait.waitedMs` | number | wall-clock ms actually spent waiting | +| `wait.timeoutMs` | number | the timeout **after clamping**, which is what was applied | +| `status` | `SessionStatus` | the session's status after the wait, so a caller that timed out still learns where things stand | +| `limitPaused` | boolean | the session is paused on a usage limit and will emit nothing until its reset, so a timeout here is expected rather than a stall worth retrying hard | Read the outcome by discriminator, in this order: @@ -390,12 +398,12 @@ read the timeout as "the worker is wedged" and kill a session that was working f ### Errors -| `errorCode` | HTTP | When | -|-------------|------|------| -| `INVALID_INPUT` | 400 | unknown `until` / `wait` token; `stop` or `blocked` requested explicitly on a mode that installs no hooks (the message names the mode); `regex=` on `/wait-output`; `match` outside 1 to 200 chars; a non-numeric `timeout` | -| `NOT_FOUND` | 404 | no such session, or one this caller does not own | -| `SESSION_BUSY` | 409 | this session's waiter cap is full | -| `RATE_LIMITED` | 429 | a per-owner or process-wide waiter cap is full. Retry later; the session you named is not the problem | +| `errorCode` | HTTP | When | +| --------------- | ---- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `INVALID_INPUT` | 400 | unknown `until` / `wait` token; `stop` or `blocked` requested explicitly on a mode that installs no hooks (the message names the mode); `regex=` on `/wait-output`; `match` outside 1 to 200 chars; a non-numeric `timeout` | +| `NOT_FOUND` | 404 | no such session, or one this caller does not own | +| `SESSION_BUSY` | 409 | this session's waiter cap is full | +| `RATE_LIMITED` | 429 | a per-owner or process-wide waiter cap is full. Retry later; the session you named is not the problem | The two capacity codes are deliberately different. A process-wide cap reported as `SESSION_BUSY` would tell the caller to switch sessions, which cannot help. The @@ -446,9 +454,9 @@ Design: [`approvals-inbox-plan.md`](approvals-inbox-plan.md). - `GET /api/v1/approvals` → `{ approvals: ApprovalItem[] }`, oldest first, ownership-scoped in multi-user mode. `ApprovalItem`: `{ id, sessionId, - sessionName, kind: 'permission'|'question'|'idle', createdAt, toolName?, - toolSummary?, message?, cwd?, context?, options?: {n, label}[], - acknowledgedAt? }`. `context` is the ANSI-stripped visible pane frame; +sessionName, kind: 'permission'|'question'|'idle', createdAt, toolName?, +toolSummary?, message?, cwd?, context?, options?: {n, label}[], +acknowledgedAt? }`. `context` is the ANSI-stripped visible pane frame; `options` is present only when the dialog's numbered choices parsed confidently; `acknowledgedAt` marks an item a human has already looked at (see `/viewed` below) and tells clients not to re-arm its tab alert. Listing @@ -466,7 +474,7 @@ Design: [`approvals-inbox-plan.md`](approvals-inbox-plan.md). first, `422 OPERATION_FAILED` when the session refused input. - `POST /api/v1/approvals/:id/dismiss` removes the item without keystrokes. - `POST /api/v1/approvals/session/:sessionId/viewed` → `{ sessionId, - acknowledged: itemId | null }`. Marks the session's pending **idle** item as +acknowledged: itemId | null }`. Marks the session's pending **idle** item as seen by a human (the web UI calls it when you open the session's tab): the item stays pending and answerable, but stops arming the yellow tab alert on every client, including after a reload. Permission/question items are never @@ -479,6 +487,47 @@ re-captured, or the item acknowledged), `approval:resolved` (`{ id, sessionId, k `resolution` one of `answered | resolved_in_terminal | superseded | session_ended | dismissed | expired`). +## Reboot restore + +A host reboot takes the tmux server down with it, so every pane dies and the +board comes up empty. At boot Codeman works out which sessions the reboot +destroyed and holds that plan in memory, and these endpoints let a client offer +it to the user. Nothing creates a pane until the user asks: the boot-time reboot +heuristic decides whether to ASK, never whether to act. + +Claude-mode sessions only (others carry their conversation id in their own +config object); remote and docker sessions are never offered, because both need +another host or container to be up. The plan is in-memory, so a server restart +drops it and the offer is gone — the conversations themselves are unaffected, +since they live in the CLI's own transcript store and stay reachable from the +Resume list. A plan nobody spends expires after 24 hours. + +- `GET /api/v1/reboot-restore` → `{ sessions: RestorableSession[], +scrollbackRestored: false }`, ownership-scoped in multi-user mode. + `RestorableSession`: `{ id, name?, workingDir, mode, owner? }`. The persisted + record itself is never sent. `scrollbackRestored` is always `false` and exists + so a client states it: a restored session is a NEW pane, so the conversation + continues and the terminal history does not. +- `POST /api/v1/reboot-restore/restore` with `{ sessionIds?: string[] }` (omit + to restore everything the caller can see) → `{ restored: RestorableSession[], +skipped: { sessionId, reason }[] }`. `reason` is one of `workspace-missing` + (the directory is gone), `workspace-forbidden` (it is outside the caller's + workspace in multi-user mode), `already-live` (the conversation is already + open, typically resumed by hand from the Resume list), `capacity-reached` + (the global or per-user session cap), or `rebuild-failed` (the agent would not + start, most often a CLI binary missing from the server's PATH). + `409 CONFLICT` when that caller already has a restore running. Entries are + removed from the plan before any pane is built, so a double-click cannot put + two panes on one conversation; anything that never became a pane goes back on + offer, except `already-live`, which cannot stop being true. A restored session + comes back attached, idle and disarmed — respawn controllers and Ralph loops + are never re-armed automatically. +- `POST /api/v1/reboot-restore/dismiss` → `{ dismissed: n }`. Drops the offer + for everything the caller can see. + +Each rebuilt session also emits the ordinary `session:created` SSE event, so +clients other than the one that clicked pick it up without refetching. + ## Read My Mind intent profiles Per-case profiles of what the user is trying to accomplish: user/agent-stated @@ -491,7 +540,7 @@ user guide: [`readmymind.md`](readmymind.md). - `GET /api/v1/sessions/:id/intent` -> `{ intent: IntentProfile }` for the session's case. `IntentProfile`: `{ key, workingDir, updatedAt, goals, - recentPrompts: { ts, sessionId, text }[] }` (prompts oldest first, FIFO cap +recentPrompts: { ts, sessionId, text }[] }` (prompts oldest first, FIFO cap 50, each <= 500 chars). A case with nothing recorded answers an empty profile with `updatedAt: 0`; nothing is persisted by reads. - `PUT /api/v1/sessions/:id/intent` with `{ goals }` (<= 8192 chars, strict @@ -524,7 +573,7 @@ same speech-to-text service the CLI's own `/voice` mode uses. Gated on the synce [`claude-voice-plan.md`](claude-voice-plan.md). - `GET /api/v1/voice/status` -> `{ available, reason?, subscriptionType?, - expiresAt? }`. `reason` is `disabled` (setting off), `no-credentials` (nobody +expiresAt? }`. `reason` is `disabled` (setting off), `no-credentials` (nobody signed in to Claude Code on the server), `expired` (the access token elapsed; running any Claude session refreshes it) or `malformed`. The OAuth token itself is never returned by this or any other endpoint. diff --git a/src/reboot-restore.ts b/src/reboot-restore.ts index c8b6bd5e..e226138c 100644 --- a/src/reboot-restore.ts +++ b/src/reboot-restore.ts @@ -57,6 +57,12 @@ export interface RebootEvidence { * * This heuristic decides whether to ASK, never whether to act. A wrong yes costs * the user a banner they dismiss, because the restore itself waits for a click. + * + * ⚠️ `os.uptime()` reports the HOST's uptime, which a container shares. A Codeman + * running in Docker therefore sees a long uptime after its own container restarts, + * the boot test fails, and no banner appears. The feature is effectively off for + * containerized installs. That is the safe direction to fail in, and fixing it + * needs a boot signal the container actually owns rather than a wider heuristic. */ export function looksLikeHostReboot(evidence: RebootEvidence): boolean { if (evidence.deadSessionCount === 0) return false; @@ -80,7 +86,14 @@ export function resolveResumeConversationId(state: SessionState): string { return chainTail || state.resumeSessionId || state.id; } -/** Why one session was passed over. Reported for logging and assertions. */ +/** + * Why one session was passed over. Reported for logging and shown to the user. + * + * The first six are decided before anything is built. `capacity-reached` and + * `rebuild-failed` can only happen once a click is spending the plan, and they + * are the two the banner must not confuse with a missing workspace: one means + * "try again after closing something", the other means the CLI would not start. + */ export interface RebootRestoreRejection { sessionId: string; reason: @@ -91,7 +104,10 @@ export interface RebootRestoreRejection { | 'unsupported-mode' | 'no-working-dir' | 'workspace-missing' - | 'already-live'; + | 'workspace-forbidden' + | 'already-live' + | 'capacity-reached' + | 'rebuild-failed'; } /** One restorable session, as the banner shows it and the rebuild replays it. */ diff --git a/src/session-env-clamp.ts b/src/session-env-clamp.ts index edaecd5c..b9abf3a6 100644 --- a/src/session-env-clamp.ts +++ b/src/session-env-clamp.ts @@ -3,10 +3,16 @@ * * A session's `envOverrides` can hand back privilege that the per-CLI config * clamp removed, so a non-granted owner's overrides get the privileged keys - * stripped before the session is built. Two callers need that today. The create - * and resume routes clamp what a request asked for, and the reboot-restore route - * clamps what a persisted record carried, because a record written while its - * owner held a grant must not replay that grant after the grant is gone. + * stripped before the session is built. The create and resume routes are what + * this bites on: they clamp what a request asked for. + * + * The reboot-restore route calls it as defence in depth, and today it can strip + * nothing. `Session.getEnvOverridesForPersist()` keeps only `CLAUDE_CODE_*` and + * `CLAUDE_CONFIG_DIR` out of a session's overrides, claude's `privilegedEnvKeys` + * are the five `ANTHROPIC_*` names, and that pass admits claude alone — so a + * persisted record cannot carry a clamped key. The call is there for the day the + * persisted set widens. The grant re-resolution that does bite on that path is + * `resolveClaudeModeForUsername`, which recomputes the permission mode. * * This lives outside `web/routes` on purpose. The question it answers is about * session privilege rather than about HTTP, and `cron/cron-service.ts` sets the diff --git a/src/web/ports/session-port.ts b/src/web/ports/session-port.ts index 61e02be3..83918763 100644 --- a/src/web/ports/session-port.ts +++ b/src/web/ports/session-port.ts @@ -4,6 +4,7 @@ */ import type { Session } from '../../session.js'; +import type { SessionState } from '../../types.js'; export interface SessionPort { readonly sessions: ReadonlyMap<string, Session>; @@ -12,5 +13,16 @@ export interface SessionPort { setupSessionListeners(session: Session): Promise<void>; persistSessionState(session: Session): void; persistSessionStateNow(session: Session): void; + /** + * Re-apply the persisted state a freshly CONSTRUCTED session does not carry: + * the pin, token and cost totals, auto-compact, auto-clear, auto-resume, nice + * priority, the flicker filter and the custom-model selection. + * + * A `Session` built from a record holds only what its constructor takes, so + * persisting it would otherwise REPLACE the fuller record with the reduced one. + * Call this before the first persist, and before `startInteractive()`, because + * the custom-model selection has to reach the pane's environment. + */ + reapplyPersistedSessionState(session: Session, saved: SessionState): Promise<void>; getSessionStateWithRespawn(session: Session): unknown; } diff --git a/src/web/public/app.js b/src/web/public/app.js index 1f0ae409..7c670497 100644 --- a/src/web/public/app.js +++ b/src/web/public/app.js @@ -957,7 +957,9 @@ class CodemanApp { this.registerServiceWorker(); // Fetch tunnel status for header indicator (desktop only) this.loadTunnelStatus(); - // Ask whether a host reboot left sessions worth rebuilding (banner, never automatic) + // Ask whether a host reboot left sessions worth rebuilding (banner, never + // automatic). handleInit() re-reads it on every SSE init; this covers the + // path where that event never arrives. this.initRebootRestoreBanner?.(); // Share a single settings fetch between both consumers const settingsPromise = fetch('/api/settings').then(r => r.ok ? r.json() : null).then(env => env?.data ?? null).catch(() => null); @@ -3761,6 +3763,12 @@ class CodemanApp { // a fresh load / reconnect (authoritative; wins over the localStorage restore). if (data.planUsage) this.updatePlanUsageChip(data.planUsage); + // A board left open across a host reboot reconnects HERE, to a server that came + // back with an empty session list. The reboot-restore offer is built at boot, + // before any client could be listening, so re-read it on every init rather than + // only on the page-load path. + this.refreshRebootRestoreBanner?.(); + // Update version displays (header and toolbar) if (data.version) { const versionEl = this.$('versionDisplay'); diff --git a/src/web/public/mobile.css b/src/web/public/mobile.css index 9be29153..d324a94f 100644 --- a/src/web/public/mobile.css +++ b/src/web/public/mobile.css @@ -3240,6 +3240,41 @@ html:is([data-skin="paper-gray"], [data-skin="solarized-light"], [data-skin="cat already reserves that space), so it needs the same safe-area padding as the other banners. The overlay is fixed and handles its own insets. ============================================================================ */ +@media (max-width: 599px) { + /* Reboot-restore banner: the same treatment as the offline banner below. Its + text and note are nowrap and the two buttons cannot shrink, so without this + the actions are pushed off a phone-width viewport and become unreachable. */ + .reboot-restore-banner { + padding: 0.4rem 0.5rem; + padding-left: calc(0.5rem + var(--safe-area-left)); + padding-right: calc(0.5rem + var(--safe-area-right)); + font-size: 0.7rem; + gap: 0.4rem; + } + + /* The session names and the scrollback note are the first things to go. The + count plus the two buttons carry the message on their own, and the note + survives as the accept button's title. */ + .reboot-restore-banner-detail, + .reboot-restore-banner-note { + display: none; + } + + .reboot-restore-banner-text { + overflow: hidden; + text-overflow: ellipsis; + } + + .reboot-restore-banner-accept, + .reboot-restore-banner-dismiss { + padding: 0.25rem 0.5rem; + } + + .reboot-restore-banner-accept { + margin-left: auto; + } +} + @media (max-width: 599px) { .offline-banner { padding: 0.4rem 0.5rem; diff --git a/src/web/public/reboot-restore-ui.js b/src/web/public/reboot-restore-ui.js index 0869eebe..aec1de81 100644 --- a/src/web/public/reboot-restore-ui.js +++ b/src/web/public/reboot-restore-ui.js @@ -8,7 +8,9 @@ * reboot guess is a heuristic and a wrong automatic restore would spawn CLI * processes nobody asked for. * - * Seeded once from `GET /api/reboot-restore` on init. Restore posts to + * Seeded from `GET /api/reboot-restore` on init and again on every SSE reconnect, + * because the tab most likely to want this is one that was open across the reboot + * and reconnects to a server that came back up with an empty board. Restore posts to * `POST /api/reboot-restore/restore`, Dismiss posts to * `POST /api/reboot-restore/dismiss`, and either way the banner goes away. The * restored sessions arrive as ordinary `session:created` events, so no extra @@ -25,6 +27,24 @@ * @loadorder 11.7 of 17, after approvals-ui.js */ +/** Plain-language wording for one skip reason, for the toast after a restore. */ +function rebootSkipReason(reason) { + switch (reason) { + case 'workspace-missing': + return 'workspace is gone'; + case 'workspace-forbidden': + return 'workspace is outside your space'; + case 'already-live': + return 'already open'; + case 'capacity-reached': + return 'session limit reached'; + case 'rebuild-failed': + return 'the agent would not start'; + default: + return reason; + } +} + Object.assign(CodemanApp.prototype, { /** Ask the server whether a reboot left anything on offer, and show the banner if so. */ async initRebootRestoreBanner() { @@ -59,6 +79,9 @@ Object.assign(CodemanApp.prototype, { detail.textContent = count > 4 ? `${names}, …` : names; detail.title = sessions.map((s) => `${s.name || s.id}\n${s.workingDir}`).join('\n\n'); } + const accept = this.$('rebootRestoreBannerAccept'); + // The note is hidden at phone width, so the warning travels on the button too. + if (accept) accept.title = 'Conversations return; terminal history does not.'; banner.hidden = false; }, @@ -66,8 +89,9 @@ Object.assign(CodemanApp.prototype, { async restoreRebootSessions() { const button = this.$('rebootRestoreBannerAccept'); if (button) button.disabled = true; - const res = await this._apiPost('/api/reboot-restore/restore', {}); - const body = res && res.ok ? await res.json().catch(() => null) : null; + // _apiJson unwraps the { success, data } envelope every /api response carries; + // reading the outer object would report every count as zero. + const body = await this._apiJson('/api/reboot-restore/restore', { method: 'POST', body: {} }); if (!body) { if (button) button.disabled = false; this.showToast?.('Could not restore the sessions', 'error'); @@ -82,10 +106,21 @@ Object.assign(CodemanApp.prototype, { this.showToast?.(`Restored ${restored} ${noun}. Terminal history did not survive the reboot.`, 'success'); } if (skipped > 0) { - this.showToast?.(`${skipped} could not be restored (workspace gone, or already open)`, 'warning'); + // Each reason means a different next step for the user, so they are not + // collapsed into one message: capacity clears by closing something, a + // failed start usually means the CLI is not on the server's PATH. + const reasons = new Set((body.skipped ?? []).map((s) => s.reason)); + this.showToast?.(`${skipped} not restored: ${[...reasons].map(rebootSkipReason).join('; ')}`, 'warning'); } }, + /** Re-read the offer after a reconnect, for a tab that was open across the reboot. */ + async refreshRebootRestoreBanner() { + const data = await this._apiJson('/api/reboot-restore'); + this._rebootRestoreSessions = data?.sessions ?? []; + this.renderRebootRestoreBanner(); + }, + /** Drop the offer. The Resume list still reaches every one of these conversations. */ async dismissRebootRestore() { this._rebootRestoreSessions = []; diff --git a/src/web/reboot-restore-registry.ts b/src/web/reboot-restore-registry.ts index e265fa62..789217eb 100644 --- a/src/web/reboot-restore-registry.ts +++ b/src/web/reboot-restore-registry.ts @@ -24,8 +24,9 @@ * - Spending is take-then-build: `take()` removes entries synchronously, before * the route's first `await`, so a double-click or two devices cannot both * reach the same entry and put two panes on one conversation. - * - One restore runs at a time. `beginSpending()` single-flights the route, so - * two concurrent clicks cannot interleave pane creation. + * - One restore runs at a time per owner. `beginSpending()` single-flights the + * route, so two concurrent clicks cannot interleave pane creation for the same + * user, while two different users never block each other. * * @dependencies reboot-restore (RebootRestoreEntry) * @consumedby web/server (plan build at boot), web/routes/reboot-restore-routes @@ -46,8 +47,13 @@ export class RebootRestoreRegistry { private entries = new Map<string, RebootRestoreEntry>(); /** When the boot pass built the plan, in ms since the epoch. */ private builtAt = 0; - /** True while a restore route call is between its take and its last pane. */ - private spending = false; + /** + * Owners with a restore in flight, between its take and its last pane. + * Keyed by owner so one user's restore does not turn another user's click into + * a conflict; `take()` already guarantees no two callers get the same entry. + * Single-user mode has one key, `undefined`, so it behaves as one global flight. + */ + private spending = new Set<string | undefined>(); /** Replace the plan with what the boot pass found. An empty list clears it. */ set(entries: readonly RebootRestoreEntry[]): void { @@ -92,9 +98,11 @@ export class RebootRestoreRegistry { /** * Put entries back after a rebuild never got as far as creating a pane. * - * Used for the click-time rejections, so a conversation the user resumed by - * hand meanwhile does not silently vanish from the banner while a workspace - * that came back stays offered. + * Used for the click-time rejections that may resolve themselves: a workspace + * that comes back, a capacity limit the user makes room under, a CLI that + * starts once its binary is on the PATH. A conversation the user resumed by + * hand is NOT put back, because that one cannot stop being true, and an entry + * the banner keeps re-offering forever is noise only Dismiss can clear. */ restore(entries: readonly RebootRestoreEntry[]): void { for (const entry of entries) this.entries.set(entry.sessionId, entry); @@ -110,24 +118,25 @@ export class RebootRestoreRegistry { } /** - * Claim the right to run a restore, or report that one is already running. - * Callers that get `true` must call `endSpending()` in a `finally`. + * Claim the right to run a restore for one owner, or report that owner already + * has one running. Callers that get `true` must call `endSpending()` in a + * `finally` with the same owner. */ - beginSpending(): boolean { - if (this.spending) return false; - this.spending = true; + beginSpending(owner?: string): boolean { + if (this.spending.has(owner)) return false; + this.spending.add(owner); return true; } - endSpending(): void { - this.spending = false; + endSpending(owner?: string): void { + this.spending.delete(owner); } /** Test hook: forget everything, including the single-flight claim. */ reset(): void { this.entries.clear(); this.builtAt = 0; - this.spending = false; + this.spending.clear(); } private dropIfExpired(): void { diff --git a/src/web/routes/reboot-restore-routes.ts b/src/web/routes/reboot-restore-routes.ts index 580b17c6..a03c74af 100644 --- a/src/web/routes/reboot-restore-routes.ts +++ b/src/web/routes/reboot-restore-routes.ts @@ -27,7 +27,14 @@ import { FastifyInstance } from 'fastify'; import { existsSync } from 'node:fs'; import { ApiErrorCode, createErrorResponse, getErrorMessage } from '../../types.js'; import { RebootRestoreRequestSchema } from '../schemas.js'; -import { parseBody, getAuthUser, canAccessOwned } from '../route-helpers.js'; +import { + parseBody, + getAuthUser, + canAccessOwned, + ownerFor, + isWorkingDirAllowed, + sessionCapacityMessage, +} from '../route-helpers.js'; import { rebootRestoreRegistry } from '../reboot-restore-registry.js'; import { rejectAlreadyLive, type RebootRestoreEntry, type RebootRestoreRejection } from '../../reboot-restore.js'; import { clampEnvOverridesForOwner } from '../../session-env-clamp.js'; @@ -76,38 +83,65 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes app.post('/api/reboot-restore/restore', async (req, reply) => { const body = parseBody(RebootRestoreRequestSchema, req.body, 'Invalid reboot restore request'); + const user = getAuthUser(req); const canAccess = accessorFor(req); + const owner = ownerFor(req); // Take BEFORE the first await: a second click must find nothing to spend. - if (!rebootRestoreRegistry.beginSpending()) { + // The flight is per owner, because `take()` already guarantees two callers + // never receive the same entry, so one user's restore need not block another's. + if (!rebootRestoreRegistry.beginSpending(owner)) { return reply.code(409).send(createErrorResponse(ApiErrorCode.CONFLICT, 'A reboot restore is already running')); } const taken = rebootRestoreRegistry.take(canAccess, body.sessionIds); + // Entries nothing built a pane for, returned to the plan on every exit path + // including a throw. Without this a failure between here and the loop would + // spend the offer and rebuild nothing, and the plan cannot be rebuilt. + const unspent = new Set(taken); try { if (taken.length === 0) return { restored: [], skipped: [] }; // The plan was built at boot and the board has moved on since. A conversation // the user resumed by hand from the Resume list is already on screen, and a - // second pane on it would fight the first for the same transcript. + // second pane on it would fight the first for the same transcript. This one + // is never re-offered: unlike a missing workspace, it cannot stop being true. const liveSessionIds = new Set(ctx.sessions.keys()); const liveConversationIds = new Set( [...ctx.sessions.values()].map((session) => session.claudeSessionId).filter((id): id is string => !!id) ); const { restore, skipped } = rejectAlreadyLive(taken, liveSessionIds, liveConversationIds); - // An entry nothing rebuilt stays on offer rather than disappearing silently. - rebootRestoreRegistry.restore(skipped.map((s) => taken.find((e) => e.sessionId === s.sessionId)!)); + for (const entry of taken) { + if (skipped.some((s) => s.sessionId === entry.sessionId)) unspent.delete(entry); + } const restored: ReturnType<typeof toBannerItem>[] = []; const failures: RebootRestoreRejection[] = [...skipped]; const workspaceHooksEnabled = await ctx.getWorkspaceHooksEnabled(); for (const entry of restore) { + // Capacity is re-checked per iteration, because this loop is itself + // creating the sessions it counts. The offer can be a day old, so the + // board may be fuller now than the plan assumed. + const capMsg = sessionCapacityMessage(ctx.sessions, entry.owner); + if (capMsg) { + failures.push({ sessionId: entry.sessionId, reason: 'capacity-reached' }); + continue; + } // A repo can be deleted between the boot that planned this and the click. if (!existsSync(entry.workingDir)) { failures.push({ sessionId: entry.sessionId, reason: 'workspace-missing' }); continue; } + // Multi-user workspace separation: the create route confines a non-admin's + // workingDir to their own case space, and a grant can be withdrawn between + // the session's creation and this restore, so the confinement is re-run + // rather than inherited from the record. + if (!isWorkingDirAllowed(user, entry.workingDir)) { + failures.push({ sessionId: entry.sessionId, reason: 'workspace-forbidden' }); + unspent.delete(entry); + continue; + } try { const saved = entry.state; const claudeModeConfig = await ctx.getClaudeModeConfig(); @@ -145,9 +179,14 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes }); await ctx.addSession(session); - ctx.persistSessionState(session); await ctx.setupSessionListeners(session); + // Before the pane spawns: the custom-model selection reaches it through + // the environment. Before the first persist: a constructed session holds + // none of this, so persisting it first would replace the fuller record + // with the reduced one and drop the pin that keeps it from being pruned. + await ctx.reapplyPersistedSessionState(session, saved); await session.startInteractive(); + ctx.persistSessionState(session); // A session without its workspace hooks goes silently blind: no stop or // idle events for respawn, no Approvals Inbox item, no red tab on a @@ -166,10 +205,22 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes ctx.broadcast(SseEvent.SessionCreated, ctx.getSessionStateWithRespawn(session)); restored.push(toBannerItem(entry)); } catch (err) { - // One workspace that has gone missing must not stop the rest of the pass. + // One entry that will not start must not stop the rest of the pass, and + // must not leave a registered session with no pane behind it: by this + // point the session is in `ctx.sessions`, holds a tab-layout slot and has + // listeners, and the commonest cause is a CLI binary that is not on the + // PATH of a freshly booted machine. console.error(`[reboot-restore] failed to rebuild ${entry.sessionId}:`, err); - failures.push({ sessionId: entry.sessionId, reason: 'workspace-missing' }); + await ctx + .cleanupSession(entry.sessionId, true, 'reboot restore failed to start the session') + .catch((cleanupErr: unknown) => + console.error(`[reboot-restore] cleanup after a failed rebuild failed: ${getErrorMessage(cleanupErr)}`) + ); + failures.push({ sessionId: entry.sessionId, reason: 'rebuild-failed' }); + // Left on offer: the user can put the binary back and click again. + continue; } + unspent.delete(entry); } if (restored.length > 0) { @@ -181,7 +232,10 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes return { restored, skipped: failures }; } finally { - rebootRestoreRegistry.endSpending(); + // Anything that never became a pane goes back on offer, including after a + // throw, so a transient failure costs a retry rather than the whole plan. + rebootRestoreRegistry.restore([...unspent]); + rebootRestoreRegistry.endSpending(owner); } }); diff --git a/src/web/server.ts b/src/web/server.ts index 065172d5..a7339328 100644 --- a/src/web/server.ts +++ b/src/web/server.ts @@ -671,6 +671,7 @@ export class WebServer extends EventEmitter { setupSessionListeners: this.setupSessionListeners.bind(this), persistSessionState: this.persistSessionState.bind(this), persistSessionStateNow: this._persistSessionStateNow.bind(this), + reapplyPersistedSessionState: this.reapplyPersistedSessionState.bind(this), getSessionStateWithRespawn: this.getSessionStateWithRespawn.bind(this), // EventPort broadcast: this.broadcast.bind(this), @@ -2907,6 +2908,53 @@ export class WebServer extends EventEmitter { return restore.length; } + /** + * Re-apply the persisted state that a `Session` constructor does not take. + * + * The reboot-restore route builds a session from a record rather than + * attaching to a surviving pane, so everything the constructor has no + * parameter for starts at its default. Persisting such a session writes + * `toState()` wholesale, which would REPLACE the record with the reduced + * version — and for a pinned session that is worse than losing a setting, + * because `cleanupSessionsByIds()` keeps a record only while it is pinned, so + * dropping the pin hands the record to the next stale sweep. + * + * Respawn and Ralph are deliberately NOT re-armed here: a machine that just + * came up is the worst moment to turn an autonomous run loose, and the user + * re-arms what they want. + */ + async reapplyPersistedSessionState(session: Session, saved: SessionState): Promise<void> { + // The custom-model env has to be rebuilt from the endpoint store: the persist + // deliberately keeps the injected VALUES out of state.json, so only the + // bookkeeping survives a restart and the values are re-derived here. + const savedCustomModel = (saved as { __customModel?: CustomModelBookkeeping }).__customModel; + if (savedCustomModel) { + session.setCustomModel(savedCustomModel, await this._rebuildCustomModelEnv(session, savedCustomModel)); + } + if (saved.pinned) session.setPinned(true); + if (saved.autoCompactEnabled !== undefined || saved.autoCompactThreshold !== undefined) { + session.setAutoCompact(saved.autoCompactEnabled ?? false, saved.autoCompactThreshold, saved.autoCompactPrompt); + } + if (saved.autoClearEnabled !== undefined || saved.autoClearThreshold !== undefined) { + session.setAutoClear(saved.autoClearEnabled ?? false, saved.autoClearThreshold); + } + if (saved.autoResumeEnabled) { + session.restoreAutoResume(true, saved.autoResumeAt); + } + if (saved.inputTokens !== undefined || saved.outputTokens !== undefined || saved.totalCost !== undefined) { + session.restoreTokens(saved.inputTokens ?? 0, saved.outputTokens ?? 0, saved.totalCost ?? 0); + // Seed the daily-usage baseline, or the restored totals are counted again as new usage. + this.lastRecordedTokens.set(session.id, { + input: saved.inputTokens ?? 0, + output: saved.outputTokens ?? 0, + }); + } + if (saved.niceEnabled !== undefined || saved.niceValue !== undefined) { + session.setNice({ enabled: saved.niceEnabled, niceValue: saved.niceValue }); + } + if (saved.flickerFilterEnabled !== undefined) session.flickerFilterEnabled = saved.flickerFilterEnabled; + } + private async restoreMuxSessions(): Promise<boolean> { try { // Reconcile mux sessions to find which ones are still alive (also discovers unknown ones) diff --git a/test/mocks/mock-route-context.ts b/test/mocks/mock-route-context.ts index 8a1bd884..401538be 100644 --- a/test/mocks/mock-route-context.ts +++ b/test/mocks/mock-route-context.ts @@ -61,6 +61,7 @@ export function createMockRouteContext(options?: { setupSessionListeners: vi.fn(async () => {}), persistSessionState: vi.fn(), persistSessionStateNow: vi.fn(), + reapplyPersistedSessionState: vi.fn(async () => {}), getSessionStateWithRespawn: vi.fn((s: MockSession) => s.toState()), // -- EventPort -- @@ -149,6 +150,7 @@ export function createMockRouteContext(options?: { clearRespawnConfig: vi.fn(), updateRespawnConfig: vi.fn(), setHistoryLimit: vi.fn(async () => {}), + startStatsCollection: vi.fn(), }, runSummaryTrackers: new Map(), activePlanOrchestrators: new Map(), diff --git a/test/routes/reboot-restore-rebuild-failure.test.ts b/test/routes/reboot-restore-rebuild-failure.test.ts new file mode 100644 index 00000000..f8f58f58 --- /dev/null +++ b/test/routes/reboot-restore-rebuild-failure.test.ts @@ -0,0 +1,210 @@ +/** + * Reboot-restore route: what happens when a rebuild gets part-way and then fails. + * + * The other route test file deliberately uses workspaces that do not exist, so it + * never reaches `new Session()`. This one mocks the `Session` module so the route + * runs its whole construction path — `addSession`, `setupSessionListeners`, + * `reapplyPersistedSessionState`, `startInteractive` — and then throws where a + * real one would when the CLI binary is missing from a freshly booted machine's + * PATH. Without the mock there is no way to exercise that path, which is how the + * original version of this route shipped a session leak the tests could not see. + * + * It also covers the session caps, because those too are only reachable once the + * route is actually willing to build something. + */ +import { describe, it, expect, afterEach, vi, beforeEach } from 'vitest'; +import Fastify, { type FastifyInstance } from 'fastify'; +import fastifyCookie from '@fastify/cookie'; + +/** Set per test: whether the mocked `startInteractive()` rejects. */ +let startShouldThrow = false; + +vi.mock('../../src/session.js', () => ({ + Session: class { + id: string; + mode: string; + name?: string; + workingDir: string; + owner?: string; + claudeSessionId: string | null = null; + constructor(config: { id: string; mode?: string; name?: string; workingDir: string; owner?: string }) { + this.id = config.id; + this.mode = config.mode ?? 'claude'; + this.name = config.name; + this.workingDir = config.workingDir; + this.owner = config.owner; + } + async startInteractive() { + if (startShouldThrow) throw new Error('spawn claude ENOENT'); + } + /** The mock route context projects a session through this on broadcast. */ + toState() { + return { id: this.id, mode: this.mode, name: this.name, workingDir: this.workingDir, owner: this.owner }; + } + }, +})); + +const { registerRebootRestoreRoutes } = await import('../../src/web/routes/reboot-restore-routes.js'); +const { rebootRestoreRegistry } = await import('../../src/web/reboot-restore-registry.js'); +const { installRouteErrorHandler } = await import('../../src/web/route-error-handler.js'); +const { httpStatusForErrorCode } = await import('../../src/types.js'); +const { createMockRouteContext } = await import('../mocks/index.js'); +type ApiErrorCode = import('../../src/types.js').ApiErrorCode; +type RebootRestoreEntry = import('../../src/reboot-restore.js').RebootRestoreEntry; +type SessionState = import('../../src/types.js').SessionState; + +/** A real directory, so the route's workspace checks pass and it reaches the build. */ +const WORKSPACE = process.cwd(); + +function offerEntry(sessionId: string, owner?: string): RebootRestoreEntry { + return { + sessionId, + name: `session ${sessionId}`, + workingDir: WORKSPACE, + owner, + mode: 'claude', + resumeConversationId: `conv-${sessionId}`, + state: { + id: sessionId, + pid: null, + status: 'idle', + workingDir: WORKSPACE, + currentTaskId: null, + createdAt: 1_760_000_000_000, + mode: 'claude', + owner, + } as SessionState, + }; +} + +async function createHarness(ctx: ReturnType<typeof createMockRouteContext>): Promise<FastifyInstance> { + const app = Fastify({ logger: false }); + await app.register(fastifyCookie); + registerRebootRestoreRoutes(app, ctx as never); + app.addHook('preSerialization', (req, reply, payload: unknown, done) => { + if (!req.url.startsWith('/api')) return done(null, payload); + if (payload === null || typeof payload !== 'object') return done(null, payload); + const p = payload as { success?: unknown; errorCode?: unknown }; + if (p.success === false) { + if (reply.statusCode === 200 && typeof p.errorCode === 'string') { + reply.code(httpStatusForErrorCode(p.errorCode as ApiErrorCode)); + } + return done(null, payload); + } + if (p.success === true) return done(null, payload); + return done(null, { success: true, data: payload }); + }); + installRouteErrorHandler(app); + await app.ready(); + return app; +} + +beforeEach(() => { + startShouldThrow = false; +}); + +afterEach(() => { + rebootRestoreRegistry.reset(); + vi.clearAllMocks(); +}); + +describe('a rebuild that fails after the session is registered', () => { + it('reports why it failed rather than blaming the workspace', async () => { + startShouldThrow = true; + rebootRestoreRegistry.set([offerEntry('a')]); + const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); + const app = await createHarness(ctx); + + const res = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data; + expect(res.restored).toEqual([]); + // Not `workspace-missing`: the directory is there, the agent would not start. + expect(res.skipped).toEqual([{ sessionId: 'a', reason: 'rebuild-failed' }]); + await app.close(); + }); + + it('does not leave a registered session with no pane behind it', async () => { + startShouldThrow = true; + rebootRestoreRegistry.set([offerEntry('a')]); + const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); + const app = await createHarness(ctx); + + await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} }); + // The session reached ctx.sessions via addSession; the route has to take it + // back out, or the board shows a tab whose pane never existed. + expect(ctx.cleanupSession).toHaveBeenCalledWith('a', true, expect.any(String)); + await app.close(); + }); + + it('keeps the entry on offer, so the user can fix the PATH and click again', async () => { + startShouldThrow = true; + rebootRestoreRegistry.set([offerEntry('a')]); + const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); + const app = await createHarness(ctx); + + await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} }); + const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data; + expect(left.sessions.map((s: { id: string }) => s.id)).toEqual(['a']); + await app.close(); + }); +}); + +describe('a rebuild that succeeds', () => { + it('re-applies the persisted state before the record is written again', async () => { + rebootRestoreRegistry.set([offerEntry('a')]); + const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); + const app = await createHarness(ctx); + + const res = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data; + expect(res.restored.map((s: { id: string }) => s.id)).toEqual(['a']); + // A session built from a record carries none of the pin, token totals or + // custom-model selection, so persisting it first would replace the fuller + // record with the reduced one. + expect(ctx.reapplyPersistedSessionState).toHaveBeenCalled(); + const reapplyOrder = (ctx.reapplyPersistedSessionState as ReturnType<typeof vi.fn>).mock.invocationCallOrder[0]; + const persistOrder = (ctx.persistSessionState as ReturnType<typeof vi.fn>).mock.invocationCallOrder[0]; + expect(reapplyOrder).toBeLessThan(persistOrder); + await app.close(); + }); + + it('tells every other board about the rebuilt session', async () => { + rebootRestoreRegistry.set([offerEntry('a')]); + const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); + const app = await createHarness(ctx); + + await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} }); + expect(ctx.broadcast).toHaveBeenCalledWith('session:created', expect.anything()); + await app.close(); + }); + + it('spends the entry, so it is no longer on offer', async () => { + rebootRestoreRegistry.set([offerEntry('a')]); + const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); + const app = await createHarness(ctx); + + await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} }); + const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data; + expect(left.sessions).toEqual([]); + await app.close(); + }); +}); + +describe('the session caps', () => { + it('stops restoring at the global cap and leaves the rest on offer', async () => { + rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]); + const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); + // Fill the board to the documented maximum of 50 concurrent sessions. + for (let i = 0; i < 50; i += 1) { + ctx.sessions.set(`filler-${i}`, { id: `filler-${i}`, owner: undefined } as never); + } + const app = await createHarness(ctx); + + const res = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data; + expect(res.restored).toEqual([]); + expect(res.skipped.map((s: { reason: string }) => s.reason)).toEqual(['capacity-reached', 'capacity-reached']); + + // Refused rather than lost: closing a session and clicking again works. + const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data; + expect(left.sessions.map((s: { id: string }) => s.id).sort()).toEqual(['a', 'b']); + await app.close(); + }); +}); diff --git a/test/routes/reboot-restore-routes.test.ts b/test/routes/reboot-restore-routes.test.ts index c16cddac..dcb7d7ed 100644 --- a/test/routes/reboot-restore-routes.test.ts +++ b/test/routes/reboot-restore-routes.test.ts @@ -23,6 +23,13 @@ import type { RebootRestoreEntry } from '../../src/reboot-restore.js'; import type { SessionState } from '../../src/types.js'; async function createHarness(authUser?: { username: string; role: 'admin' | 'user' }): Promise<FastifyInstance> { + return createHarnessWithCtx(createMockRouteContext(), authUser); +} + +async function createHarnessWithCtx( + ctx: ReturnType<typeof createMockRouteContext>, + authUser?: { username: string; role: 'admin' | 'user' } +): Promise<FastifyInstance> { const app = Fastify({ logger: false }); await app.register(fastifyCookie); if (authUser) { @@ -30,7 +37,7 @@ async function createHarness(authUser?: { username: string; role: 'admin' | 'use (req as unknown as { authUser: typeof authUser }).authUser = authUser; }); } - registerRebootRestoreRoutes(app, createMockRouteContext() as never); + registerRebootRestoreRoutes(app, ctx as never); app.addHook('preSerialization', (req, reply, payload: unknown, done) => { if (!req.url.startsWith('/api')) return done(null, payload); @@ -106,18 +113,36 @@ describe('GET /api/reboot-restore', () => { }); describe('POST /api/reboot-restore/restore', () => { - it('spends the offer, so a second click finds nothing left to spend', async () => { + it('reports a workspace that is gone, and keeps offering it in case it comes back', async () => { rebootRestoreRegistry.set([offerEntry('a')]); const app = await createHarness(); const first = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data; - // The workspace is gone, so nothing was rebuilt — but the entry was taken. expect(first.restored).toEqual([]); expect(first.skipped).toEqual([{ sessionId: 'a', reason: 'workspace-missing' }]); - const second = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data; - expect(second.restored).toEqual([]); - expect(second.skipped).toEqual([]); + // Nothing was built, so the entry goes back: a repo can be restored from a + // backup between two clicks, and losing the offer would be unrecoverable. + const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data; + expect(left.sessions.map((s: { id: string }) => s.id)).toEqual(['a']); + await app.close(); + }); + + it('never re-offers a conversation that is already open', async () => { + const entry = offerEntry('a'); + rebootRestoreRegistry.set([entry]); + const app = await createHarness(); + const ctx = createMockRouteContext({ sessionId: entry.sessionId }); + // A session with that id is live, which is what the Resume list would produce. + const liveApp = await createHarnessWithCtx(ctx); + + const res = (await liveApp.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data; + expect(res.skipped).toEqual([{ sessionId: 'a', reason: 'already-live' }]); + + // Unlike a missing workspace, this one is dropped: it cannot stop being true. + const left = (await liveApp.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data; + expect(left.sessions).toEqual([]); + await liveApp.close(); await app.close(); }); @@ -132,8 +157,9 @@ describe('POST /api/reboot-restore/restore', () => { }); expect(res.json().data.skipped).toEqual([{ sessionId: 'b', reason: 'workspace-missing' }]); + // 'a' was never taken, and 'b' came back because no pane was built for it. const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data; - expect(left.sessions.map((s: { id: string }) => s.id)).toEqual(['a']); + expect(left.sessions.map((s: { id: string }) => s.id).sort()).toEqual(['a', 'b']); await app.close(); }); @@ -150,12 +176,13 @@ describe('POST /api/reboot-restore/restore', () => { it('turns a second concurrent restore away rather than interleaving it', async () => { rebootRestoreRegistry.set([offerEntry('a')]); - // Claimed by a restore already in flight. - expect(rebootRestoreRegistry.beginSpending()).toBe(true); + // Claimed by a restore already in flight for this same owner (undefined in + // single-user mode, which is what the harness runs as). + expect(rebootRestoreRegistry.beginSpending(undefined)).toBe(true); const app = await createHarness(); const res = await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} }); expect(res.statusCode).toBe(409); - rebootRestoreRegistry.endSpending(); + rebootRestoreRegistry.endSpending(undefined); await app.close(); }); }); From fa52753e8bb0a33996c8fd774c101b917a7acf94 Mon Sep 17 00:00:00 2001 From: Michael Grundberg <michael.grundberg@irisity.com> Date: Wed, 16 Sep 2026 15:04:35 +0200 Subject: [PATCH 12/28] fix(sessions): undo a failed rebuild without deleting the user's data A second review of the previous commit found that its own repair for the session leak introduced three defects, all from reaching for cleanupSession() to undo a half-built session. That function is the user-initiated delete, not an undo. It banked the session's historical token and cost totals into the lifetime figures, and a reboot never runs cleanup, so those totals had never been counted before; every failed rebuild added them again. It saw the pin that had just been restored and demoted the record to `stopped`, which this pass reads as the durable marker of a deliberate kill, so a pinned session whose rebuild failed became permanently unrestorable. And it recursively removed `.claude-images` from the working directory, which belongs to the workspace rather than to the session, so a failed rebuild destroyed the pasted images of any other live session in that repo. discardPartiallyBuiltSession() now undoes only what the construction did: the map entry, the tab-layout slot, the listeners and any pane the launch created before throwing. The persisted record, the lifetime totals, the Ralph state and the workspace's files are left alone. Re-applying the persisted state also splits in two, which removes the first two defects at the root rather than only at the call site. The half that shapes the pane, the custom-model environment and the nice priority, still runs before the spawn. The half that is the session's own history now runs after it, so a session whose pane never started carries no totals and no pin for anything downstream to misread. The rest of that review. The multi-user workspace confinement re-check read the requesting user's grant, and returns true for an admin, so the case its own comment described was the one it missed; it now resolves the entry owner's grant through isWorkingDirAllowedForUsername, the way cron does. A forbidden workspace goes back on offer, matching both the registry's stated contract and the API reference. The client re-reads the plan after a restore instead of blanking the banner, so entries the server put back stay reachable, and a 409 now says a restore is already running rather than reporting a failure. A dismiss arriving mid-restore wins, through a generation counter the route carries across its take. The re-application also restores the tab colour, the image-watcher flag and the original pinnedAt, via a new Session.restorePin that does not re-stamp the pin time. The phone breakpoint gains min-width: 0, without which a nowrap flex item never shrinks and the buttons still overflow, and it folds into the existing phone block. Ralph's loop configuration still does not survive a restore, because toState() reads it off a live tracker and there is no way to keep it without arming the loop. The method now says so rather than leaving it implied. Tests. The capacity test could not fail on the property it existed for: it filled the board past the cap before the loop, so a single pre-loop check would have passed it. It now leaves one seat, so only a per-iteration check restores exactly one entry. New tests cover the ordering around the spawn, a throw before the loop returning the whole plan and releasing the flight, the dismiss-during-restore race, and that the failure path calls the narrow discard rather than the delete. The shared mock context gains the port method it was missing, which is what made the first run of these tests fail for the wrong reason. Refs #411 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- src/session.ts | 13 +++ src/web/ports/session-port.ts | 24 +++-- src/web/public/mobile.css | 6 +- src/web/public/reboot-restore-ui.js | 22 +++-- src/web/reboot-restore-registry.ts | 21 +++- src/web/routes/reboot-restore-routes.ts | 43 +++++--- src/web/server.ts | 91 ++++++++++++++--- test/mocks/mock-route-context.ts | 3 + .../reboot-restore-rebuild-failure.test.ts | 97 ++++++++++++++++++- 9 files changed, 273 insertions(+), 47 deletions(-) diff --git a/src/session.ts b/src/session.ts index 21d5ac89..16720e59 100644 --- a/src/session.ts +++ b/src/session.ts @@ -1499,6 +1499,19 @@ export class Session extends EventEmitter { this._pinnedAt = pinned ? Date.now() : null; } + /** + * Restore a pin from a persisted record, keeping the moment it was pinned. + * + * `setPinned()` stamps `pinnedAt` with now, which is right for a user pinning a + * session and wrong for a restore: the session-manager orders its pinned group + * by that stamp, so a restored session would jump to the front of a list it had + * been sitting further down. + */ + restorePin(pinned: boolean, pinnedAt?: number): void { + this._pinned = pinned; + this._pinnedAt = pinned ? (pinnedAt ?? Date.now()) : null; + } + get flickerFilterEnabled(): boolean { return this._flickerFilterEnabled; } diff --git a/src/web/ports/session-port.ts b/src/web/ports/session-port.ts index 83918763..85add1d8 100644 --- a/src/web/ports/session-port.ts +++ b/src/web/ports/session-port.ts @@ -14,15 +14,27 @@ export interface SessionPort { persistSessionState(session: Session): void; persistSessionStateNow(session: Session): void; /** - * Re-apply the persisted state a freshly CONSTRUCTED session does not carry: - * the pin, token and cost totals, auto-compact, auto-clear, auto-resume, nice - * priority, the flicker filter and the custom-model selection. + * Re-apply the persisted state a freshly CONSTRUCTED session does not carry. * * A `Session` built from a record holds only what its constructor takes, so * persisting it would otherwise REPLACE the fuller record with the reduced one. - * Call this before the first persist, and before `startInteractive()`, because - * the custom-model selection has to reach the pane's environment. + * Two phases: `before-spawn` shapes the pane (the custom-model environment and + * the nice priority) and must precede `startInteractive()`; `after-spawn` is + * the session's own history (the pin, token and cost totals, auto-compact, + * auto-clear, auto-resume, colour, image watcher, flicker filter) and must NOT + * land on a session whose pane failed to start. */ - reapplyPersistedSessionState(session: Session, saved: SessionState): Promise<void>; + reapplyPersistedSessionState( + session: Session, + saved: SessionState, + phase: 'before-spawn' | 'after-spawn' + ): Promise<void>; + /** + * Undo a session that was registered but never got a working pane: the map + * entry, its tab-layout slot, and any pane the launch created before throwing. + * Unlike {@link cleanupSession} it leaves the persisted record, the lifetime + * token totals, the Ralph state and the workspace's own files untouched. + */ + discardPartiallyBuiltSession(sessionId: string): Promise<void>; getSessionStateWithRespawn(session: Session): unknown; } diff --git a/src/web/public/mobile.css b/src/web/public/mobile.css index d324a94f..f364c701 100644 --- a/src/web/public/mobile.css +++ b/src/web/public/mobile.css @@ -3260,7 +3260,11 @@ html:is([data-skin="paper-gray"], [data-skin="solarized-light"], [data-skin="cat display: none; } + /* A flex item will not shrink below its content width at the default + `min-width: auto`, so without this the nowrap text pushes the buttons off a + 360px viewport and the ellipsis never engages. */ .reboot-restore-banner-text { + min-width: 0; overflow: hidden; text-overflow: ellipsis; } @@ -3273,9 +3277,7 @@ html:is([data-skin="paper-gray"], [data-skin="solarized-light"], [data-skin="cat .reboot-restore-banner-accept { margin-left: auto; } -} -@media (max-width: 599px) { .offline-banner { padding: 0.4rem 0.5rem; padding-left: calc(0.5rem + var(--safe-area-left)); diff --git a/src/web/public/reboot-restore-ui.js b/src/web/public/reboot-restore-ui.js index aec1de81..1cdc8da8 100644 --- a/src/web/public/reboot-restore-ui.js +++ b/src/web/public/reboot-restore-ui.js @@ -23,7 +23,7 @@ * * @mixin Extends CodemanApp.prototype via Object.assign * @dependency app.js (CodemanApp class, showToast) - * @dependency api-client.js at runtime (this._apiJson / this._apiPost) + * @dependency api-client.js at runtime (this._api / this._apiJson) * @loadorder 11.7 of 17, after approvals-ui.js */ @@ -89,9 +89,15 @@ Object.assign(CodemanApp.prototype, { async restoreRebootSessions() { const button = this.$('rebootRestoreBannerAccept'); if (button) button.disabled = true; - // _apiJson unwraps the { success, data } envelope every /api response carries; - // reading the outer object would report every count as zero. - const body = await this._apiJson('/api/reboot-restore/restore', { method: 'POST', body: {} }); + const res = await this._api('/api/reboot-restore/restore', { method: 'POST', body: {} }); + if (res && res.status === 409) { + if (button) button.disabled = false; + this.showToast?.('A restore is already running', 'info'); + return; + } + // The uniform envelope wraps every /api payload; reading the outer object + // would report every count as zero. + const body = res && res.ok ? (await res.json().catch(() => null))?.data : null; if (!body) { if (button) button.disabled = false; this.showToast?.('Could not restore the sessions', 'error'); @@ -99,8 +105,12 @@ Object.assign(CodemanApp.prototype, { } const restored = body.restored?.length ?? 0; const skipped = body.skipped?.length ?? 0; - this._rebootRestoreSessions = []; - this.renderRebootRestoreBanner(); + // Re-read rather than clearing: the server puts back anything it could not + // build for a reason that may pass, such as a session limit or an agent that + // would not start, and blanking the banner here would put those entries out + // of reach until a reload. + await this.refreshRebootRestoreBanner(); + if (button) button.disabled = false; if (restored > 0) { const noun = restored === 1 ? 'conversation' : 'conversations'; this.showToast?.(`Restored ${restored} ${noun}. Terminal history did not survive the reboot.`, 'success'); diff --git a/src/web/reboot-restore-registry.ts b/src/web/reboot-restore-registry.ts index 789217eb..28b4a1d7 100644 --- a/src/web/reboot-restore-registry.ts +++ b/src/web/reboot-restore-registry.ts @@ -47,6 +47,13 @@ export class RebootRestoreRegistry { private entries = new Map<string, RebootRestoreEntry>(); /** When the boot pass built the plan, in ms since the epoch. */ private builtAt = 0; + /** + * Bumped by anything that invalidates entries a restore is already holding. + * A Dismiss arriving mid-restore must win: without this the route's `finally` + * would put its unspent entries back and resurrect the offer the user just + * cleared, with a fresh 24-hour life. + */ + private generation = 0; /** * Owners with a restore in flight, between its take and its last pane. * Keyed by owner so one user's restore does not turn another user's click into @@ -59,6 +66,12 @@ export class RebootRestoreRegistry { set(entries: readonly RebootRestoreEntry[]): void { this.entries = new Map(entries.map((entry) => [entry.sessionId, entry])); this.builtAt = entries.length > 0 ? Date.now() : 0; + this.generation += 1; + } + + /** The current generation, for a caller that will later return entries. */ + currentGeneration(): number { + return this.generation; } /** @@ -104,7 +117,10 @@ export class RebootRestoreRegistry { * hand is NOT put back, because that one cannot stop being true, and an entry * the banner keeps re-offering forever is noise only Dismiss can clear. */ - restore(entries: readonly RebootRestoreEntry[]): void { + restore(entries: readonly RebootRestoreEntry[], generation?: number): void { + // A dismiss (or a fresh boot plan) since the caller took these entries means + // they are no longer wanted back. + if (generation !== undefined && generation !== this.generation) return; for (const entry of entries) this.entries.set(entry.sessionId, entry); if (entries.length > 0 && this.builtAt === 0) this.builtAt = Date.now(); } @@ -114,6 +130,8 @@ export class RebootRestoreRegistry { const removable = [...this.entries.values()].filter((entry) => canAccess(entry.owner)); for (const entry of removable) this.entries.delete(entry.sessionId); if (this.entries.size === 0) this.builtAt = 0; + // Any restore currently in flight must not put its entries back afterwards. + this.generation += 1; return removable.length; } @@ -137,6 +155,7 @@ export class RebootRestoreRegistry { this.entries.clear(); this.builtAt = 0; this.spending.clear(); + this.generation += 1; } private dropIfExpired(): void { diff --git a/src/web/routes/reboot-restore-routes.ts b/src/web/routes/reboot-restore-routes.ts index a03c74af..636f2cd5 100644 --- a/src/web/routes/reboot-restore-routes.ts +++ b/src/web/routes/reboot-restore-routes.ts @@ -32,7 +32,7 @@ import { getAuthUser, canAccessOwned, ownerFor, - isWorkingDirAllowed, + isWorkingDirAllowedForUsername, sessionCapacityMessage, } from '../route-helpers.js'; import { rebootRestoreRegistry } from '../reboot-restore-registry.js'; @@ -83,7 +83,6 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes app.post('/api/reboot-restore/restore', async (req, reply) => { const body = parseBody(RebootRestoreRequestSchema, req.body, 'Invalid reboot restore request'); - const user = getAuthUser(req); const canAccess = accessorFor(req); const owner = ownerFor(req); @@ -93,6 +92,7 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes if (!rebootRestoreRegistry.beginSpending(owner)) { return reply.code(409).send(createErrorResponse(ApiErrorCode.CONFLICT, 'A reboot restore is already running')); } + const generation = rebootRestoreRegistry.currentGeneration(); const taken = rebootRestoreRegistry.take(canAccess, body.sessionIds); // Entries nothing built a pane for, returned to the plan on every exit path // including a throw. Without this a failure between here and the loop would @@ -136,10 +136,15 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes // Multi-user workspace separation: the create route confines a non-admin's // workingDir to their own case space, and a grant can be withdrawn between // the session's creation and this restore, so the confinement is re-run - // rather than inherited from the record. - if (!isWorkingDirAllowed(user, entry.workingDir)) { + // rather than inherited from the record. Keyed on the OWNER, not on the + // caller: an admin spending another user's entry must be held to that + // user's confinement, and `isWorkingDirAllowed` would wave an admin + // through. The same reason the two grant re-checks below read + // `saved.owner`. + if (!(await isWorkingDirAllowedForUsername(entry.owner, entry.workingDir))) { + // Left on offer: a withdrawn grant can be restored, unlike an already-open + // conversation, so this is not the permanent kind of refusal. failures.push({ sessionId: entry.sessionId, reason: 'workspace-forbidden' }); - unspent.delete(entry); continue; } try { @@ -180,12 +185,16 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes await ctx.addSession(session); await ctx.setupSessionListeners(session); - // Before the pane spawns: the custom-model selection reaches it through - // the environment. Before the first persist: a constructed session holds - // none of this, so persisting it first would replace the fuller record - // with the reduced one and drop the pin that keeps it from being pruned. - await ctx.reapplyPersistedSessionState(session, saved); + // Shapes the pane, so it has to land before the CLI process starts. + await ctx.reapplyPersistedSessionState(session, saved, 'before-spawn'); await session.startInteractive(); + // The session's own history, applied only once the pane exists: on a + // failed start these totals would belong to a session that never ran. + // Both halves precede the first persist, because a constructed session + // carries none of this and `toState()` is written wholesale, so + // persisting first would replace the fuller record with the reduced one + // and drop the pin that keeps it from being pruned. + await ctx.reapplyPersistedSessionState(session, saved, 'after-spawn'); ctx.persistSessionState(session); // A session without its workspace hooks goes silently blind: no stop or @@ -211,10 +220,15 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes // listeners, and the commonest cause is a CLI binary that is not on the // PATH of a freshly booted machine. console.error(`[reboot-restore] failed to rebuild ${entry.sessionId}:`, err); + // Not cleanupSession(): that is the user-initiated delete, and it would + // count this session's historical tokens into the lifetime totals, demote + // a pinned record to `stopped` (which this pass reads as an intentional + // kill, making the session permanently unrestorable) and delete the + // workspace's `.claude-images`. This undoes only the construction. await ctx - .cleanupSession(entry.sessionId, true, 'reboot restore failed to start the session') - .catch((cleanupErr: unknown) => - console.error(`[reboot-restore] cleanup after a failed rebuild failed: ${getErrorMessage(cleanupErr)}`) + .discardPartiallyBuiltSession(entry.sessionId) + .catch((discardErr: unknown) => + console.error(`[reboot-restore] discarding a failed rebuild failed: ${getErrorMessage(discardErr)}`) ); failures.push({ sessionId: entry.sessionId, reason: 'rebuild-failed' }); // Left on offer: the user can put the binary back and click again. @@ -234,7 +248,8 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes } finally { // Anything that never became a pane goes back on offer, including after a // throw, so a transient failure costs a retry rather than the whole plan. - rebootRestoreRegistry.restore([...unspent]); + // Passing the generation makes a Dismiss that landed mid-restore win. + rebootRestoreRegistry.restore([...unspent], generation); rebootRestoreRegistry.endSpending(owner); } }); diff --git a/src/web/server.ts b/src/web/server.ts index a7339328..474d8cd2 100644 --- a/src/web/server.ts +++ b/src/web/server.ts @@ -672,6 +672,7 @@ export class WebServer extends EventEmitter { persistSessionState: this.persistSessionState.bind(this), persistSessionStateNow: this._persistSessionStateNow.bind(this), reapplyPersistedSessionState: this.reapplyPersistedSessionState.bind(this), + discardPartiallyBuiltSession: this.discardPartiallyBuiltSession.bind(this), getSessionStateWithRespawn: this.getSessionStateWithRespawn.bind(this), // EventPort broadcast: this.broadcast.bind(this), @@ -2919,19 +2920,42 @@ export class WebServer extends EventEmitter { * because `cleanupSessionsByIds()` keeps a record only while it is pinned, so * dropping the pin hands the record to the next stale sweep. * - * Respawn and Ralph are deliberately NOT re-armed here: a machine that just - * came up is the worst moment to turn an autonomous run loose, and the user - * re-arms what they want. + * Split in two phases because the two halves have opposite timing needs: + * + * - `before-spawn` shapes the pane itself, so it has to land before the CLI + * process starts. The custom-model selection is an environment injection and + * the nice priority is applied to the spawn. + * - `after-spawn` is the session's own accumulated history. It must NOT land + * on a session whose pane failed to start: the totals would then belong to a + * session that never ran, and any later cleanup would add them to the + * lifetime figures a second time. + * + * Respawn and Ralph are deliberately NOT re-armed: a machine that just came up + * is the worst moment to turn an autonomous run loose, and the user re-arms + * what they want. Ralph's loop CONFIGURATION does not survive either, because + * `toState()` reads `ralphEnabled` and the completion phrase off a live + * tracker, and there is no way to hold them without arming the loop. */ - async reapplyPersistedSessionState(session: Session, saved: SessionState): Promise<void> { - // The custom-model env has to be rebuilt from the endpoint store: the persist - // deliberately keeps the injected VALUES out of state.json, so only the - // bookkeeping survives a restart and the values are re-derived here. - const savedCustomModel = (saved as { __customModel?: CustomModelBookkeeping }).__customModel; - if (savedCustomModel) { - session.setCustomModel(savedCustomModel, await this._rebuildCustomModelEnv(session, savedCustomModel)); + async reapplyPersistedSessionState( + session: Session, + saved: SessionState, + phase: 'before-spawn' | 'after-spawn' + ): Promise<void> { + if (phase === 'before-spawn') { + // The custom-model env has to be rebuilt from the endpoint store: the persist + // deliberately keeps the injected VALUES out of state.json, so only the + // bookkeeping survives a restart and the values are re-derived here. + const savedCustomModel = (saved as { __customModel?: CustomModelBookkeeping }).__customModel; + if (savedCustomModel) { + session.setCustomModel(savedCustomModel, await this._rebuildCustomModelEnv(session, savedCustomModel)); + } + if (saved.niceEnabled !== undefined || saved.niceValue !== undefined) { + session.setNice({ enabled: saved.niceEnabled, niceValue: saved.niceValue }); + } + return; } - if (saved.pinned) session.setPinned(true); + + if (saved.pinned) session.restorePin(true, saved.pinnedAt); if (saved.autoCompactEnabled !== undefined || saved.autoCompactThreshold !== undefined) { session.setAutoCompact(saved.autoCompactEnabled ?? false, saved.autoCompactThreshold, saved.autoCompactPrompt); } @@ -2949,12 +2973,51 @@ export class WebServer extends EventEmitter { output: saved.outputTokens ?? 0, }); } - if (saved.niceEnabled !== undefined || saved.niceValue !== undefined) { - session.setNice({ enabled: saved.niceEnabled, niceValue: saved.niceValue }); - } + if (saved.color) session.setColor(saved.color); + if (saved.imageWatcherEnabled !== undefined) session.imageWatcherEnabled = saved.imageWatcherEnabled; if (saved.flickerFilterEnabled !== undefined) session.flickerFilterEnabled = saved.flickerFilterEnabled; } + /** + * Undo a session that was registered but never got a working pane. + * + * Deliberately NOT `cleanupSession()`, which is the user-initiated delete: that + * path adds the session's token totals to the lifetime figures, demotes a + * pinned record to `stopped` (the durable marker of an intentional kill, which + * would make the session permanently ineligible for a reboot restore), drops + * the persisted Ralph state, and recursively removes `.claude-images` from the + * WORKING DIRECTORY, which belongs to the workspace rather than to this session + * and may hold another live session's pasted images. + * + * This undoes only what the failed construction did: the map entry, the tab + * layout slot `registerSessionWithLayout()` took, and any pane the CLI launch + * managed to create before it threw. The persisted record is left exactly as it + * was, so the session stays restorable on the next attempt. + */ + async discardPartiallyBuiltSession(sessionId: string): Promise<void> { + const session = this.sessions.get(sessionId); + if (!session) return; + this.sessions.delete(sessionId); + this.sse.cleanupSessionBatches(sessionId); + this.persistDeb.cancelKey(sessionId); + try { + session.removeAllListeners(); + await session.stop?.(); + } catch (err) { + console.warn(`[Server] stopping a partially built session failed: ${getErrorMessage(err)}`); + } + try { + await this.mux.killSession(sessionId); + } catch { + // The pane may never have been created; nothing to kill is the normal case. + } + try { + await this.tabLayouts.sessionsRemoved([{ id: sessionId, owner: session.owner }]); + } catch (err) { + console.warn(`[Server] releasing the tab layout slot failed: ${getErrorMessage(err)}`); + } + } + private async restoreMuxSessions(): Promise<boolean> { try { // Reconcile mux sessions to find which ones are still alive (also discovers unknown ones) diff --git a/test/mocks/mock-route-context.ts b/test/mocks/mock-route-context.ts index 401538be..9cb26d6a 100644 --- a/test/mocks/mock-route-context.ts +++ b/test/mocks/mock-route-context.ts @@ -62,6 +62,9 @@ export function createMockRouteContext(options?: { persistSessionState: vi.fn(), persistSessionStateNow: vi.fn(), reapplyPersistedSessionState: vi.fn(async () => {}), + discardPartiallyBuiltSession: vi.fn(async (id: string) => { + sessions.delete(id); + }), getSessionStateWithRespawn: vi.fn((s: MockSession) => s.toState()), // -- EventPort -- diff --git a/test/routes/reboot-restore-rebuild-failure.test.ts b/test/routes/reboot-restore-rebuild-failure.test.ts index f8f58f58..f1a35417 100644 --- a/test/routes/reboot-restore-rebuild-failure.test.ts +++ b/test/routes/reboot-restore-rebuild-failure.test.ts @@ -18,6 +18,8 @@ import fastifyCookie from '@fastify/cookie'; /** Set per test: whether the mocked `startInteractive()` rejects. */ let startShouldThrow = false; +/** Ordering log, so a test can assert what ran before the pane spawned. */ +const callOrder: string[] = []; vi.mock('../../src/session.js', () => ({ Session: class { @@ -35,6 +37,7 @@ vi.mock('../../src/session.js', () => ({ this.owner = config.owner; } async startInteractive() { + callOrder.push('startInteractive'); if (startShouldThrow) throw new Error('spawn claude ENOENT'); } /** The mock route context projects a session through this on broadcast. */ @@ -101,6 +104,7 @@ async function createHarness(ctx: ReturnType<typeof createMockRouteContext>): Pr beforeEach(() => { startShouldThrow = false; + callOrder.length = 0; }); afterEach(() => { @@ -131,7 +135,12 @@ describe('a rebuild that fails after the session is registered', () => { await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} }); // The session reached ctx.sessions via addSession; the route has to take it // back out, or the board shows a tab whose pane never existed. - expect(ctx.cleanupSession).toHaveBeenCalledWith('a', true, expect.any(String)); + expect(ctx.discardPartiallyBuiltSession).toHaveBeenCalledWith('a'); + expect(ctx.sessions.has('a')).toBe(false); + // NOT the user-initiated delete: that would bank this session's historical + // tokens into the lifetime totals, demote a pinned record to `stopped`, and + // delete the workspace's .claude-images. + expect(ctx.cleanupSession).not.toHaveBeenCalled(); await app.close(); }); @@ -166,6 +175,23 @@ describe('a rebuild that succeeds', () => { await app.close(); }); + it('shapes the pane before it spawns, and restores the history after', async () => { + rebootRestoreRegistry.set([offerEntry('a')]); + const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); + (ctx.reapplyPersistedSessionState as ReturnType<typeof vi.fn>).mockImplementation( + async (_s: unknown, _saved: unknown, phase: string) => { + callOrder.push(`reapply:${phase}`); + } + ); + const app = await createHarness(ctx); + + await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} }); + // The custom-model environment has to reach the process; the token totals + // must not land on a session whose pane never started. + expect(callOrder).toEqual(['reapply:before-spawn', 'startInteractive', 'reapply:after-spawn']); + await app.close(); + }); + it('tells every other board about the rebuilt session', async () => { rebootRestoreRegistry.set([offerEntry('a')]); const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); @@ -189,10 +215,30 @@ describe('a rebuild that succeeds', () => { }); describe('the session caps', () => { - it('stops restoring at the global cap and leaves the rest on offer', async () => { + it('counts the sessions it is itself creating, not just the ones it started with', async () => { + rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]); + const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); + // One seat short of the documented maximum of 50, counting the session the + // mock context seeds. A check that ran once before the loop would restore + // BOTH entries; only a per-iteration check refuses the second. + for (let i = 0; i < 48; i += 1) { + ctx.sessions.set(`filler-${i}`, { id: `filler-${i}`, owner: undefined } as never); + } + const app = await createHarness(ctx); + + const res = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data; + expect(res.restored.map((s: { id: string }) => s.id)).toEqual(['a']); + expect(res.skipped).toEqual([{ sessionId: 'b', reason: 'capacity-reached' }]); + + // Refused rather than lost: closing a session and clicking again works. + const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data; + expect(left.sessions.map((s: { id: string }) => s.id)).toEqual(['b']); + await app.close(); + }); + + it('refuses every entry when the board is already at the cap', async () => { rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]); const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); - // Fill the board to the documented maximum of 50 concurrent sessions. for (let i = 0; i < 50; i += 1) { ctx.sessions.set(`filler-${i}`, { id: `filler-${i}`, owner: undefined } as never); } @@ -201,10 +247,53 @@ describe('the session caps', () => { const res = (await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} })).json().data; expect(res.restored).toEqual([]); expect(res.skipped.map((s: { reason: string }) => s.reason)).toEqual(['capacity-reached', 'capacity-reached']); + await app.close(); + }); +}); - // Refused rather than lost: closing a session and clicking again works. +describe('a failure before any entry is considered', () => { + it('returns the whole plan rather than spending it', async () => { + rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]); + const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); + (ctx.getWorkspaceHooksEnabled as ReturnType<typeof vi.fn>).mockRejectedValue(new Error('settings unreadable')); + const app = await createHarness(ctx); + + const res = await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} }); + expect(res.statusCode).toBeGreaterThanOrEqual(500); + + // The plan cannot be rebuilt once boot has pruned the records, so a throw + // anywhere in the route has to hand the entries back. const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data; expect(left.sessions.map((s: { id: string }) => s.id).sort()).toEqual(['a', 'b']); await app.close(); }); + + it('releases the single flight, so the next click is not refused', async () => { + rebootRestoreRegistry.set([offerEntry('a')]); + const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); + (ctx.getWorkspaceHooksEnabled as ReturnType<typeof vi.fn>).mockRejectedValue(new Error('settings unreadable')); + const app = await createHarness(ctx); + + await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} }); + expect(rebootRestoreRegistry.beginSpending(undefined)).toBe(true); + rebootRestoreRegistry.endSpending(undefined); + await app.close(); + }); +}); + +describe('a dismiss that lands while a restore is running', () => { + it('wins, rather than being undone when the restore hands its entries back', async () => { + const entries = [offerEntry('a')]; + rebootRestoreRegistry.set(entries); + const generation = rebootRestoreRegistry.currentGeneration(); + const taken = rebootRestoreRegistry.take(() => true); + expect(taken).toHaveLength(1); + + // The user clears the banner while the restore is still working. + rebootRestoreRegistry.clear(() => true); + // The restore finishes and tries to put its unspent entry back. + rebootRestoreRegistry.restore(taken, generation); + + expect(rebootRestoreRegistry.list(() => true)).toEqual([]); + }); }); From 71ed7b127c0ce4f14c1f1a71a9abc3ee21fe1c50 Mon Sep 17 00:00:00 2001 From: Michael Grundberg <michael.grundberg@irisity.com> Date: Wed, 16 Sep 2026 15:34:47 +0200 Subject: [PATCH 13/28] fix(sessions): make the discard a real inverse of the construction MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Third review of the reboot-restore branch. The narrow discard the previous commit introduced avoided everything cleanupSession() did wrongly, and in dropping so much of it also dropped four things it had to keep. The worst broke the retry the whole design rests on. setupSessionListeners() returns early while sessionListenerRefs still holds the session id, and the discard never cleared that entry. So the advertised flow — a rebuild fails because the agent binary is missing, the user fixes their PATH and clicks again — reused the same id, wired no listeners at all, and produced a tab that never showed output, never updated its status and never persisted. That is worse than the leak the discard was added to prevent. Three more registrations leaked with it: a RunSummaryTracker and its interval, an image watcher on the workspace, and the Ralph fix-plan watcher. The discard now undoes each registration setupSessionListeners() makes, in its order, and the per-session custom-model config directory, which holds the endpoint's API key literally and which nothing else would ever remove. The image-watcher flag was restored after the code that reads it, so a session came back reporting the feature as on with nothing watching. It moves to the before-spawn phase, and that phase now runs before the listeners rather than after them. The generation counter that lets a mid-restore dismiss win was global while clear() is ownership-scoped, so one user's dismiss discarded another user's unspent entries, permanently, because nothing rebuilds an in-memory plan. It is now per owner. Bumping only the owners of entries the dismiss removed was not enough either: take() has already emptied the plan by then, so a dismiss landing mid-restore saw nothing of that owner's to remove and invalidated nothing. The owners that matter are those with a restore in flight, filtered by what the dismissing user may access, and that is what clear() now bumps. Plan expiry bumps too, so a restore straddling the 24-hour boundary cannot hand entries back and give an expired plan another full day. Tests. discardPartiallyBuiltSession had no test at all: the only implementation any test ran was the mock's one-line stub, which is why every defect above was invisible. test/discard-partially-built-session.ts drives the real WebServer, and the retry assertion fails if the listener refs are left behind — verified by reverting the fix. The dismiss-race test drove the registry by hand, so deleting the route's generation argument left it green; it now goes through the route, and two further tests cover the multi-user cases. The mock context has now gone stale twice, because route tests pass it as `ctx as never` and tsconfig.json includes only src, so nothing ever compares it to the ports. A type-level guard is therefore inert — I wrote one and confirmed it never fires. test/mocks/mock-route-context-completeness.ts compares the mock's keys against WebServer.createRouteContext() at runtime instead, and names what is missing. Also: the API reference now says workspace-forbidden is judged against the owner's grant, the banner's module header no longer claims Restore always dismisses it, and the detail span gets the same min-width: 0 the phone rule already needed. Refs #411 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- docs/api-reference.md | 5 +- src/web/public/reboot-restore-ui.js | 10 +- src/web/public/styles.css | 4 + src/web/reboot-restore-registry.ts | 73 ++++++++--- src/web/routes/reboot-restore-routes.ts | 23 ++-- src/web/server.ts | 58 +++++++-- test/discard-partially-built-session.test.ts | 113 ++++++++++++++++++ .../mock-route-context-completeness.test.ts | 28 +++++ .../reboot-restore-rebuild-failure.test.ts | 64 ++++++++-- 9 files changed, 322 insertions(+), 56 deletions(-) create mode 100644 test/discard-partially-built-session.test.ts create mode 100644 test/mocks/mock-route-context-completeness.test.ts diff --git a/docs/api-reference.md b/docs/api-reference.md index 85349ee3..4bca9768 100644 --- a/docs/api-reference.md +++ b/docs/api-reference.md @@ -511,8 +511,9 @@ scrollbackRestored: false }`, ownership-scoped in multi-user mode. - `POST /api/v1/reboot-restore/restore` with `{ sessionIds?: string[] }` (omit to restore everything the caller can see) → `{ restored: RestorableSession[], skipped: { sessionId, reason }[] }`. `reason` is one of `workspace-missing` - (the directory is gone), `workspace-forbidden` (it is outside the caller's - workspace in multi-user mode), `already-live` (the conversation is already + (the directory is gone), `workspace-forbidden` (in multi-user mode it is + outside the workspace of the user the session belongs to, re-checked against + that owner's current grant rather than the caller's), `already-live` (the conversation is already open, typically resumed by hand from the Resume list), `capacity-reached` (the global or per-user session cap), or `rebuild-failed` (the agent would not start, most often a CLI binary missing from the server's PATH). diff --git a/src/web/public/reboot-restore-ui.js b/src/web/public/reboot-restore-ui.js index 1cdc8da8..2a3599dd 100644 --- a/src/web/public/reboot-restore-ui.js +++ b/src/web/public/reboot-restore-ui.js @@ -11,10 +11,12 @@ * Seeded from `GET /api/reboot-restore` on init and again on every SSE reconnect, * because the tab most likely to want this is one that was open across the reboot * and reconnects to a server that came back up with an empty board. Restore posts to - * `POST /api/reboot-restore/restore`, Dismiss posts to - * `POST /api/reboot-restore/dismiss`, and either way the banner goes away. The - * restored sessions arrive as ordinary `session:created` events, so no extra - * rendering is needed here. + * `POST /api/reboot-restore/restore` and Dismiss posts to + * `POST /api/reboot-restore/dismiss`. Dismiss always clears the banner; Restore + * re-reads the plan afterwards, because the server puts back anything it could + * not build for a reason that may pass, such as a session limit or an agent that + * would not start. The restored sessions arrive as ordinary `session:created` + * events, so no extra rendering is needed here. * * The banner says that terminal history did not survive, because a restored * session is a new pane: the conversation continues and the scrollback does not. diff --git a/src/web/public/styles.css b/src/web/public/styles.css index b5e4873b..f35bf5cd 100644 --- a/src/web/public/styles.css +++ b/src/web/public/styles.css @@ -15277,6 +15277,10 @@ html[data-skin="daylight-blue"] .welcome-btn-tunnel.active:hover { .reboot-restore-banner-detail { color: rgba(255, 255, 255, 0.8); font-weight: 500; + /* A flex item will not shrink below its content width at the default + `min-width: auto`, so without this the session names push the buttons out of + the line between the phone breakpoint and full width. */ + min-width: 0; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; diff --git a/src/web/reboot-restore-registry.ts b/src/web/reboot-restore-registry.ts index 28b4a1d7..0dfab538 100644 --- a/src/web/reboot-restore-registry.ts +++ b/src/web/reboot-restore-registry.ts @@ -48,12 +48,16 @@ export class RebootRestoreRegistry { /** When the boot pass built the plan, in ms since the epoch. */ private builtAt = 0; /** - * Bumped by anything that invalidates entries a restore is already holding. - * A Dismiss arriving mid-restore must win: without this the route's `finally` - * would put its unspent entries back and resurrect the offer the user just - * cleared, with a fresh 24-hour life. + * Per owner, bumped by anything that invalidates that owner's entries while a + * restore is already holding them. A Dismiss arriving mid-restore must win: + * without this the route's `finally` would put its unspent entries back and + * resurrect the offer the user just cleared, with a fresh 24-hour life. + * + * Keyed by owner rather than global, because `clear()` is ownership-scoped. A + * single counter would let one user's Dismiss discard another user's unspent + * entries, and the plan is in-memory, so those offers would be gone for good. */ - private generation = 0; + private generations = new Map<string | undefined, number>(); /** * Owners with a restore in flight, between its take and its last pane. * Keyed by owner so one user's restore does not turn another user's click into @@ -66,12 +70,17 @@ export class RebootRestoreRegistry { set(entries: readonly RebootRestoreEntry[]): void { this.entries = new Map(entries.map((entry) => [entry.sessionId, entry])); this.builtAt = entries.length > 0 ? Date.now() : 0; - this.generation += 1; + this.bumpAll(); } - /** The current generation, for a caller that will later return entries. */ - currentGeneration(): number { - return this.generation; + /** + * The generations of the owners of `entries`, for a caller that will hand some + * of them back later. Pass the result to {@link restore}. + */ + snapshotGenerations(entries: readonly RebootRestoreEntry[]): Map<string | undefined, number> { + const snapshot = new Map<string | undefined, number>(); + for (const entry of entries) snapshot.set(entry.owner, this.generations.get(entry.owner) ?? 0); + return snapshot; } /** @@ -117,12 +126,20 @@ export class RebootRestoreRegistry { * hand is NOT put back, because that one cannot stop being true, and an entry * the banner keeps re-offering forever is noise only Dismiss can clear. */ - restore(entries: readonly RebootRestoreEntry[], generation?: number): void { - // A dismiss (or a fresh boot plan) since the caller took these entries means - // they are no longer wanted back. - if (generation !== undefined && generation !== this.generation) return; - for (const entry of entries) this.entries.set(entry.sessionId, entry); - if (entries.length > 0 && this.builtAt === 0) this.builtAt = Date.now(); + restore(entries: readonly RebootRestoreEntry[], generations?: ReadonlyMap<string | undefined, number>): void { + let added = 0; + for (const entry of entries) { + // A dismiss (or a fresh boot plan) for THIS entry's owner since the caller + // took it means it is no longer wanted back. Another owner's dismiss is + // none of this entry's business. + if (generations) { + const taken = generations.get(entry.owner); + if (taken !== undefined && taken !== (this.generations.get(entry.owner) ?? 0)) continue; + } + this.entries.set(entry.sessionId, entry); + added += 1; + } + if (added > 0 && this.builtAt === 0) this.builtAt = Date.now(); } /** Drop the entries a viewer can see. Returns how many went. */ @@ -130,8 +147,14 @@ export class RebootRestoreRegistry { const removable = [...this.entries.values()].filter((entry) => canAccess(entry.owner)); for (const entry of removable) this.entries.delete(entry.sessionId); if (this.entries.size === 0) this.builtAt = 0; - // Any restore currently in flight must not put its entries back afterwards. - this.generation += 1; + // A restore in flight for these owners must not put their entries back. The + // in-flight owners are the ones that matter and the ones the plan can no + // longer name: `take()` has already removed their entries, so a dismiss that + // lands mid-restore sees nothing of theirs to remove. The bump is limited to + // owners this caller could see, so it cannot reach anyone else's restore. + const invalidated = new Set(removable.map((entry) => entry.owner)); + for (const owner of this.spending) if (canAccess(owner)) invalidated.add(owner); + for (const owner of invalidated) this.bump(owner); return removable.length; } @@ -155,11 +178,25 @@ export class RebootRestoreRegistry { this.entries.clear(); this.builtAt = 0; this.spending.clear(); - this.generation += 1; + this.generations.clear(); + } + + private bump(owner: string | undefined): void { + this.generations.set(owner, (this.generations.get(owner) ?? 0) + 1); + } + + /** Invalidate every owner's in-flight returns, including owners not yet seen. */ + private bumpAll(): void { + for (const owner of new Set([...this.entries.values()].map((entry) => entry.owner))) this.bump(owner); + for (const owner of [...this.generations.keys()]) this.bump(owner); } private dropIfExpired(): void { if (this.builtAt > 0 && Date.now() - this.builtAt > PLAN_TTL_MS) { + // Bump before clearing, while the owners are still known: a restore that + // took entries just before the expiry must not hand them back afterwards + // and give an expired plan another full day of life. + this.bumpAll(); this.entries.clear(); this.builtAt = 0; } diff --git a/src/web/routes/reboot-restore-routes.ts b/src/web/routes/reboot-restore-routes.ts index 636f2cd5..da742cc4 100644 --- a/src/web/routes/reboot-restore-routes.ts +++ b/src/web/routes/reboot-restore-routes.ts @@ -92,8 +92,8 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes if (!rebootRestoreRegistry.beginSpending(owner)) { return reply.code(409).send(createErrorResponse(ApiErrorCode.CONFLICT, 'A reboot restore is already running')); } - const generation = rebootRestoreRegistry.currentGeneration(); const taken = rebootRestoreRegistry.take(canAccess, body.sessionIds); + const generations = rebootRestoreRegistry.snapshotGenerations(taken); // Entries nothing built a pane for, returned to the plan on every exit path // including a throw. Without this a failure between here and the loop would // spend the offer and rebuild nothing, and the plan cannot be rebuilt. @@ -184,16 +184,20 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes }); await ctx.addSession(session); - await ctx.setupSessionListeners(session); - // Shapes the pane, so it has to land before the CLI process starts. + // Before the listeners, because setupSessionListeners() reads the + // image-watcher flag this phase restores; before the spawn, because the + // custom-model environment and the nice priority shape the process. await ctx.reapplyPersistedSessionState(session, saved, 'before-spawn'); + await ctx.setupSessionListeners(session); await session.startInteractive(); // The session's own history, applied only once the pane exists: on a // failed start these totals would belong to a session that never ran. - // Both halves precede the first persist, because a constructed session - // carries none of this and `toState()` is written wholesale, so - // persisting first would replace the fuller record with the reduced one - // and drop the pin that keeps it from being pruned. + // Both halves precede the route's OWN persist, which matters because a + // constructed session carries none of this and `toState()` is written + // wholesale, so persisting first would replace the fuller record with + // the reduced one and drop the pin that keeps it from being pruned. A + // listener-driven persist can still land inside the debounce window + // while the pane starts; the write below repairs the record. await ctx.reapplyPersistedSessionState(session, saved, 'after-spawn'); ctx.persistSessionState(session); @@ -248,8 +252,9 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes } finally { // Anything that never became a pane goes back on offer, including after a // throw, so a transient failure costs a retry rather than the whole plan. - // Passing the generation makes a Dismiss that landed mid-restore win. - rebootRestoreRegistry.restore([...unspent], generation); + // Passing the generations makes a Dismiss that landed mid-restore win, for + // the owners it actually covered. + rebootRestoreRegistry.restore([...unspent], generations); rebootRestoreRegistry.endSpending(owner); } }); diff --git a/src/web/server.ts b/src/web/server.ts index 474d8cd2..095b094d 100644 --- a/src/web/server.ts +++ b/src/web/server.ts @@ -2923,8 +2923,9 @@ export class WebServer extends EventEmitter { * Split in two phases because the two halves have opposite timing needs: * * - `before-spawn` shapes the pane itself, so it has to land before the CLI - * process starts. The custom-model selection is an environment injection and - * the nice priority is applied to the spawn. + * process starts, and before `setupSessionListeners()`, which reads the + * image-watcher flag. The custom-model selection is an environment injection + * and the nice priority is applied to the spawn. * - `after-spawn` is the session's own accumulated history. It must NOT land * on a session whose pane failed to start: the totals would then belong to a * session that never ran, and any later cleanup would add them to the @@ -2952,6 +2953,10 @@ export class WebServer extends EventEmitter { if (saved.niceEnabled !== undefined || saved.niceValue !== undefined) { session.setNice({ enabled: saved.niceEnabled, niceValue: saved.niceValue }); } + // `setupSessionListeners()` READS this flag to decide whether to start the + // watcher, so setting it later would leave the session reporting the feature + // as on with nothing watching. + if (saved.imageWatcherEnabled !== undefined) session.imageWatcherEnabled = saved.imageWatcherEnabled; return; } @@ -2974,7 +2979,6 @@ export class WebServer extends EventEmitter { }); } if (saved.color) session.setColor(saved.color); - if (saved.imageWatcherEnabled !== undefined) session.imageWatcherEnabled = saved.imageWatcherEnabled; if (saved.flickerFilterEnabled !== undefined) session.flickerFilterEnabled = saved.flickerFilterEnabled; } @@ -2989,33 +2993,61 @@ export class WebServer extends EventEmitter { * WORKING DIRECTORY, which belongs to the workspace rather than to this session * and may hold another live session's pasted images. * - * This undoes only what the failed construction did: the map entry, the tab - * layout slot `registerSessionWithLayout()` took, and any pane the CLI launch - * managed to create before it threw. The persisted record is left exactly as it - * was, so the session stays restorable on the next attempt. + * Everything else `_doCleanupSession()` does, this has to do as well. It is the + * inverse of `registerSessionWithLayout()` plus `setupSessionListeners()`, and + * every registration those two make has to come back out — above all + * `sessionListenerRefs`, whose presence makes `setupSessionListeners()` return + * early. Leaving that entry behind is worse than the leak this function exists + * to prevent: the retry reuses the same session id, wires no listeners at all, + * and the user gets a tab that never shows output. + * + * The persisted record, the lifetime totals, the stored Ralph state and the + * workspace's own files are left exactly as they were, so the session stays + * restorable on the next attempt. */ async discardPartiallyBuiltSession(sessionId: string): Promise<void> { const session = this.sessions.get(sessionId); if (!session) return; this.sessions.delete(sessionId); + + // --- the inverse of setupSessionListeners(), in its order --- + const summaryTracker = this.runSummaryTrackers.get(sessionId); + if (summaryTracker) { + summaryTracker.stop(); + this.runSummaryTrackers.delete(sessionId); + } + // An fs.watch on the workspace (or on @fix_plan.md) that nothing else closes. + session.ralphTracker.stopWatchingFixPlan(); + // An FSWatcher on the workspace, likewise. + imageWatcher.unwatchSession(sessionId); + const listeners = this.sessionListenerRefs.get(sessionId); + if (listeners) { + detachSessionListeners(session, listeners); + this.sessionListenerRefs.delete(sessionId); + } + + // --- the inverse of the construction itself --- this.sse.cleanupSessionBatches(sessionId); this.persistDeb.cancelKey(sessionId); + fileStreamManager.closeSessionStreams(sessionId); + // The per-session custom-model config dir carries the endpoint's API key, and + // `before-spawn` may already have written it. Nothing else would ever remove + // it: the stale sweep only touches state.json. A retry rewrites it. + removeConfigDir(customModelConfigDir(sessionId)); try { session.removeAllListeners(); - await session.stop?.(); + await session.stop(true); } catch (err) { console.warn(`[Server] stopping a partially built session failed: ${getErrorMessage(err)}`); } - try { - await this.mux.killSession(sessionId); - } catch { - // The pane may never have been created; nothing to kill is the normal case. - } try { await this.tabLayouts.sessionsRemoved([{ id: sessionId, owner: session.owner }]); } catch (err) { console.warn(`[Server] releasing the tab layout slot failed: ${getErrorMessage(err)}`); } + // Any `session:updated` the half-built session emitted before it failed left a + // tab on every other open board, and the client's handler is an upsert. + this.broadcast(SseEvent.SessionDeleted, { id: sessionId }); } private async restoreMuxSessions(): Promise<boolean> { diff --git a/test/discard-partially-built-session.test.ts b/test/discard-partially-built-session.test.ts new file mode 100644 index 00000000..7232e969 --- /dev/null +++ b/test/discard-partially-built-session.test.ts @@ -0,0 +1,113 @@ +/** + * `WebServer.discardPartiallyBuiltSession()` against the real server object. + * + * The reboot-restore route calls this when a rebuild registers a session and + * then fails to start its pane. It has to be the exact inverse of + * `registerSessionWithLayout()` plus `setupSessionListeners()`, and it must NOT + * be the user-initiated delete: banking the session's token totals, demoting a + * pinned record or deleting the workspace's files would all be wrong for a + * session that never ran. + * + * These tests drive the real method rather than the route, because the route + * tests run against a mock context whose `discardPartiallyBuiltSession` is a + * one-line stub — an earlier version of this function left four registrations + * behind and every route test still passed. + * + * The retry assertion is the important one. `setupSessionListeners()` returns + * early when `sessionListenerRefs` still holds the session id, so a discard that + * leaves that entry makes the next attempt wire nothing at all, and the user + * gets a tab that never shows output. + */ +import { mkdirSync, rmSync } from 'node:fs'; +import { homedir } from 'node:os'; +import { join } from 'node:path'; +import { afterEach, beforeEach, describe, expect, it } from 'vitest'; + +import { WebServer } from '../src/web/server.js'; +import { Session } from '../src/session.js'; +import { TmuxManager } from '../src/tmux-manager.js'; + +/** Reach the private collections the discard is responsible for emptying. */ +interface ServerInternals { + sessions: Map<string, Session>; + sessionListenerRefs: Map<string, unknown>; + runSummaryTrackers: Map<string, unknown>; + registerSessionWithLayout(session: Session): Promise<void>; + setupSessionListeners(session: Session): Promise<void>; + discardPartiallyBuiltSession(sessionId: string): Promise<void>; +} + +const WORKSPACE = join(homedir(), '.codeman-test-discard'); +const SESSION_ID = 'a1b2c3d4e5f60718'; + +let server: WebServer; +let internals: ServerInternals; +let mux: TmuxManager; + +function buildSession(): Session { + return new Session({ + id: SESSION_ID, + workingDir: WORKSPACE, + mode: 'claude', + name: 'rebuilt session', + mux, + useMux: true, + }); +} + +beforeEach(() => { + mkdirSync(WORKSPACE, { recursive: true }); + // Test mode: no port is opened and no CLI is launched. + server = new WebServer(0, false, true); + internals = server as unknown as ServerInternals; + mux = new TmuxManager(); +}); + +afterEach(async () => { + await internals.discardPartiallyBuiltSession(SESSION_ID).catch(() => {}); + rmSync(WORKSPACE, { recursive: true, force: true }); +}); + +describe('discarding a session whose pane never started', () => { + it('takes the session back out of the server', async () => { + const session = buildSession(); + await internals.registerSessionWithLayout(session); + await internals.setupSessionListeners(session); + expect(internals.sessions.has(SESSION_ID)).toBe(true); + + await internals.discardPartiallyBuiltSession(SESSION_ID); + expect(internals.sessions.has(SESSION_ID)).toBe(false); + }); + + it('releases the listener registration, so a retry can wire itself again', async () => { + const first = buildSession(); + await internals.registerSessionWithLayout(first); + await internals.setupSessionListeners(first); + expect(internals.sessionListenerRefs.has(SESSION_ID)).toBe(true); + + await internals.discardPartiallyBuiltSession(SESSION_ID); + expect(internals.sessionListenerRefs.has(SESSION_ID)).toBe(false); + + // The retry reuses the id by design. `setupSessionListeners()` returns early + // while the refs are still there, so a session built now would run blind: + // no terminal output, no status updates, no exit broadcast. + const retry = buildSession(); + await internals.registerSessionWithLayout(retry); + await internals.setupSessionListeners(retry); + expect(internals.sessionListenerRefs.has(SESSION_ID)).toBe(true); + }); + + it('stops the run-summary tracker, whose interval would otherwise keep firing', async () => { + const session = buildSession(); + await internals.registerSessionWithLayout(session); + await internals.setupSessionListeners(session); + expect(internals.runSummaryTrackers.has(SESSION_ID)).toBe(true); + + await internals.discardPartiallyBuiltSession(SESSION_ID); + expect(internals.runSummaryTrackers.has(SESSION_ID)).toBe(false); + }); + + it('does nothing at all for a session it never registered', async () => { + await expect(internals.discardPartiallyBuiltSession('never-existed')).resolves.toBeUndefined(); + }); +}); diff --git a/test/mocks/mock-route-context-completeness.test.ts b/test/mocks/mock-route-context-completeness.test.ts new file mode 100644 index 00000000..effb98c8 --- /dev/null +++ b/test/mocks/mock-route-context-completeness.test.ts @@ -0,0 +1,28 @@ +/** + * The mock route context must offer everything the real one does. + * + * Route tests pass their context as `ctx as never`, and `tsconfig.json` includes + * only `src/**`, so no type check ever compares the mock against the ports. A + * port that gained a method left this mock missing it twice; both times the + * route under test threw a TypeError inside its own catch, and the suite + * reported a plausible-looking failure for an unrelated reason. + * + * So the comparison is made at runtime, against `WebServer.createRouteContext()` + * rather than against the port types, which is what keeps it from drifting: the + * server's own context object is the thing route modules are really given. + */ +import { describe, expect, it } from 'vitest'; + +import { WebServer } from '../../src/web/server.js'; +import { createMockRouteContext } from './mock-route-context.js'; + +describe('the mock route context', () => { + it('offers every member the real route context does', () => { + const server = new WebServer(0, false, true); + const real = (server as unknown as { createRouteContext(): Record<string, unknown> }).createRouteContext(); + const mock = createMockRouteContext() as unknown as Record<string, unknown>; + + const missing = Object.keys(real).filter((key) => !(key in mock)); + expect(missing, `mock-route-context.ts is missing: ${missing.join(', ')}`).toEqual([]); + }); +}); diff --git a/test/routes/reboot-restore-rebuild-failure.test.ts b/test/routes/reboot-restore-rebuild-failure.test.ts index f1a35417..9ee93fb4 100644 --- a/test/routes/reboot-restore-rebuild-failure.test.ts +++ b/test/routes/reboot-restore-rebuild-failure.test.ts @@ -282,18 +282,62 @@ describe('a failure before any entry is considered', () => { }); describe('a dismiss that lands while a restore is running', () => { - it('wins, rather than being undone when the restore hands its entries back', async () => { - const entries = [offerEntry('a')]; - rebootRestoreRegistry.set(entries); - const generation = rebootRestoreRegistry.currentGeneration(); - const taken = rebootRestoreRegistry.take(() => true); - expect(taken).toHaveLength(1); + it('wins, rather than being undone when the route hands its entries back', async () => { + rebootRestoreRegistry.set([offerEntry('a')]); + const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); + // The user clicks Dismiss while the restore is between its take and its + // return. Driven through the ROUTE, so removing the generation argument from + // the route would make this fail. + (ctx.getWorkspaceHooksEnabled as ReturnType<typeof vi.fn>).mockImplementation(async () => { + rebootRestoreRegistry.clear(() => true); + throw new Error('settings unreadable'); + }); + const app = await createHarness(ctx); - // The user clears the banner while the restore is still working. - rebootRestoreRegistry.clear(() => true); - // The restore finishes and tries to put its unspent entry back. - rebootRestoreRegistry.restore(taken, generation); + await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} }); + + const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data; + expect(left.sessions).toEqual([]); + await app.close(); + }); + + it('reaches an in-flight restore the dismisser can see, even once its entries are taken', async () => { + const mine = offerEntry('mine', 'alice'); + rebootRestoreRegistry.set([mine]); + expect(rebootRestoreRegistry.beginSpending('alice')).toBe(true); + const generations = rebootRestoreRegistry.snapshotGenerations([mine]); + const taken = rebootRestoreRegistry.take((owner) => owner === 'alice'); + + // The plan is empty now, so a dismiss has nothing of Alice's to remove; the + // invalidation has to come from her claimed flight. + rebootRestoreRegistry.clear((owner) => owner === 'alice'); + rebootRestoreRegistry.restore(taken, generations); + rebootRestoreRegistry.endSpending('alice'); expect(rebootRestoreRegistry.list(() => true)).toEqual([]); }); + + it('does not reach another owner, whose unspent entries still come back', async () => { + const mine = offerEntry('mine', 'alice'); + const theirs = offerEntry('theirs', 'bob'); + rebootRestoreRegistry.set([mine, theirs]); + + // Bob is mid-restore, holding his own entry. The claimed flight is what makes + // this the interesting case: a dismiss can no longer see Bob's entries in the + // plan, so the invalidation has to come from the in-flight set, filtered by + // what the dismissing user may access. + expect(rebootRestoreRegistry.beginSpending('bob')).toBe(true); + const bobsGenerations = rebootRestoreRegistry.snapshotGenerations([theirs]); + const bobsTaken = rebootRestoreRegistry.take((owner) => owner === 'bob'); + expect(bobsTaken.map((e) => e.sessionId)).toEqual(['theirs']); + + // Alice dismisses her own banner meanwhile. + rebootRestoreRegistry.clear((owner) => owner === 'alice'); + + // Bob's restore finishes and hands his entry back. Alice's dismiss covered + // her entries, not his, so his offer survives. + rebootRestoreRegistry.restore(bobsTaken, bobsGenerations); + rebootRestoreRegistry.endSpending('bob'); + expect(rebootRestoreRegistry.list(() => true).map((e) => e.sessionId)).toEqual(['theirs']); + }); }); From 39976041e0032c1295c08559b1e3627b4c1d16b7 Mon Sep 17 00:00:00 2001 From: Michael Grundberg <michael.grundberg@irisity.com> Date: Wed, 16 Sep 2026 16:51:00 +0200 Subject: [PATCH 14/28] fix(sessions): let a dismiss reach the entries a restore is holding MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fourth review of the reboot-restore branch, and the third to find a defect in the previous round's fix. This one is the same shape as its predecessor: a counter keyed on one thing, compared against a set keyed on another. The generation counter was indexed by the entry's owner, while the in-flight set holds the caller doing the restoring. Those are the same person exactly when a user restores their own sessions, which is every case the tests covered. The route deliberately supports the other case: an admin may spend another user's entries. So when an admin restored Bob's sessions and Bob dismissed the banner, nothing matched, the entries came back, and a plan Bob had explicitly dismissed was re-armed for another twenty-four hours. Rather than reconcile the two key spaces, the counter is gone. `take()` now parks the entries it hands out, remembering which caller is spending them, and they stay parked until that restore ends. A dismiss filters the parked entries by `canAccess(entry.owner)` — the same predicate it already applies to the plan — so it reaches them wherever they are. `releaseFlight()` puts back only what is still parked. Expiry and a fresh boot plan unpark everything, for the same reason. There is one key space now, the entry's owner, and the spender is only ever used to tell two concurrent flights apart. That removes `generations`, `snapshotGenerations()`, `bump()`, `bumpAll()` and the argument threaded through the route. The discard grew the teardown it still lacked. A rebuild can fail after startInteractive() resolved, and a restored workspace still carries Codeman's hooks, so the CLI can post a hook event within milliseconds; the transcript watcher that starts from it, the attachment registry, the wait registry and the approvals inbox all outlive the listeners and would meet the retry, which reuses the session id by design. Its steps also run in reverse order now, so no live listener can reach a tracker that has already stopped, and the mux kill has its own guard, because stop() kills the pane in its last block after destroying four trackers. Tests. The run-summary test named an interval and asserted a map entry, so dropping stop() left it green; it now spies on stop(). Nothing pinned that before-spawn must precede setupSessionListeners, which reads the flag that phase restores, so swapping the two lines was silent; the ordering test now includes the listener setup. The retry assertion was a tautology and now asserts a different refs object. Both strengthened tests were verified by reverting their fix. Two new tests cover the admin-restores-another-owner cases this round was about. The server in the discard test is built once and stopped, since its constructor registers handlers on module-level watchers, and the workspace is removed through safeRmHomeTree. Refs #411 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- src/web/reboot-restore-registry.ts | 99 +++++++++---------- src/web/routes/reboot-restore-routes.ts | 9 +- src/web/server.ts | 39 ++++++-- test/discard-partially-built-session.test.ts | 36 +++++-- test/reboot-restore.test.ts | 12 +-- .../reboot-restore-rebuild-failure.test.ts | 67 +++++++++---- 6 files changed, 159 insertions(+), 103 deletions(-) diff --git a/src/web/reboot-restore-registry.ts b/src/web/reboot-restore-registry.ts index 0dfab538..94a7c63f 100644 --- a/src/web/reboot-restore-registry.ts +++ b/src/web/reboot-restore-registry.ts @@ -48,16 +48,18 @@ export class RebootRestoreRegistry { /** When the boot pass built the plan, in ms since the epoch. */ private builtAt = 0; /** - * Per owner, bumped by anything that invalidates that owner's entries while a - * restore is already holding them. A Dismiss arriving mid-restore must win: - * without this the route's `finally` would put its unspent entries back and - * resurrect the offer the user just cleared, with a fresh 24-hour life. + * Entries handed to a restore that has not finished, by session id, each + * remembering which caller is spending it. * - * Keyed by owner rather than global, because `clear()` is ownership-scoped. A - * single counter would let one user's Dismiss discard another user's unspent - * entries, and the plan is in-memory, so those offers would be gone for good. + * A taken entry is still part of the offer until its restore resolves it, so + * it has to stay reachable by everything that can invalidate an offer. Holding + * the entries themselves — rather than a counter to compare against later — + * means `clear()` filters them by the SAME `canAccess(entry.owner)` predicate + * it already applies to the plan. A counter cannot do that, because the caller + * spending an entry need not be its owner: an admin may restore another user's + * sessions, and then the spender and the owner are different keys. */ - private generations = new Map<string | undefined, number>(); + private parked = new Map<string, { entry: RebootRestoreEntry; spender: string | undefined }>(); /** * Owners with a restore in flight, between its take and its last pane. * Keyed by owner so one user's restore does not turn another user's click into @@ -70,17 +72,8 @@ export class RebootRestoreRegistry { set(entries: readonly RebootRestoreEntry[]): void { this.entries = new Map(entries.map((entry) => [entry.sessionId, entry])); this.builtAt = entries.length > 0 ? Date.now() : 0; - this.bumpAll(); - } - - /** - * The generations of the owners of `entries`, for a caller that will hand some - * of them back later. Pass the result to {@link restore}. - */ - snapshotGenerations(entries: readonly RebootRestoreEntry[]): Map<string | undefined, number> { - const snapshot = new Map<string | undefined, number>(); - for (const entry of entries) snapshot.set(entry.owner, this.generations.get(entry.owner) ?? 0); - return snapshot; + // A fresh boot plan supersedes anything an in-flight restore still holds. + this.parked.clear(); } /** @@ -104,7 +97,11 @@ export class RebootRestoreRegistry { * * @param sessionIds The ids to spend, or undefined for every visible entry. */ - take(canAccess: (owner: string | undefined) => boolean, sessionIds?: readonly string[]): RebootRestoreEntry[] { + take( + canAccess: (owner: string | undefined) => boolean, + sessionIds: readonly string[] | undefined, + spender: string | undefined + ): RebootRestoreEntry[] { this.dropIfExpired(); const wanted = sessionIds ? new Set(sessionIds) : undefined; const taken: RebootRestoreEntry[] = []; @@ -112,6 +109,9 @@ export class RebootRestoreRegistry { if (wanted && !wanted.has(entry.sessionId)) continue; if (!canAccess(entry.owner)) continue; this.entries.delete(entry.sessionId); + // Parked rather than forgotten: until this restore resolves the entry, a + // dismiss still has to be able to reach and cancel it. + this.parked.set(entry.sessionId, { entry, spender }); taken.push(entry); } return taken; @@ -126,18 +126,19 @@ export class RebootRestoreRegistry { * hand is NOT put back, because that one cannot stop being true, and an entry * the banner keeps re-offering forever is noise only Dismiss can clear. */ - restore(entries: readonly RebootRestoreEntry[], generations?: ReadonlyMap<string | undefined, number>): void { + releaseFlight(spender: string | undefined, keep: readonly RebootRestoreEntry[]): void { + const wanted = new Set(keep.map((entry) => entry.sessionId)); let added = 0; - for (const entry of entries) { - // A dismiss (or a fresh boot plan) for THIS entry's owner since the caller - // took it means it is no longer wanted back. Another owner's dismiss is - // none of this entry's business. - if (generations) { - const taken = generations.get(entry.owner); - if (taken !== undefined && taken !== (this.generations.get(entry.owner) ?? 0)) continue; + for (const [sessionId, held] of [...this.parked]) { + if (held.spender !== spender) continue; + this.parked.delete(sessionId); + // Still parked means nothing cancelled it while the restore ran. A dismiss, + // an expiry or a fresh boot plan removes it from `parked`, and then it does + // not come back however the restore ended. + if (wanted.has(sessionId)) { + this.entries.set(sessionId, held.entry); + added += 1; } - this.entries.set(entry.sessionId, entry); - added += 1; } if (added > 0 && this.builtAt === 0) this.builtAt = Date.now(); } @@ -146,16 +147,17 @@ export class RebootRestoreRegistry { clear(canAccess: (owner: string | undefined) => boolean): number { const removable = [...this.entries.values()].filter((entry) => canAccess(entry.owner)); for (const entry of removable) this.entries.delete(entry.sessionId); + // Entries a restore is holding are dismissed by the same rule, so a dismiss + // that lands mid-restore wins. Judged on the ENTRY's owner, exactly as above, + // rather than on who happens to be restoring it. + let parkedRemoved = 0; + for (const [sessionId, held] of [...this.parked]) { + if (!canAccess(held.entry.owner)) continue; + this.parked.delete(sessionId); + parkedRemoved += 1; + } if (this.entries.size === 0) this.builtAt = 0; - // A restore in flight for these owners must not put their entries back. The - // in-flight owners are the ones that matter and the ones the plan can no - // longer name: `take()` has already removed their entries, so a dismiss that - // lands mid-restore sees nothing of theirs to remove. The bump is limited to - // owners this caller could see, so it cannot reach anyone else's restore. - const invalidated = new Set(removable.map((entry) => entry.owner)); - for (const owner of this.spending) if (canAccess(owner)) invalidated.add(owner); - for (const owner of invalidated) this.bump(owner); - return removable.length; + return removable.length + parkedRemoved; } /** @@ -176,27 +178,16 @@ export class RebootRestoreRegistry { /** Test hook: forget everything, including the single-flight claim. */ reset(): void { this.entries.clear(); + this.parked.clear(); this.builtAt = 0; this.spending.clear(); - this.generations.clear(); - } - - private bump(owner: string | undefined): void { - this.generations.set(owner, (this.generations.get(owner) ?? 0) + 1); - } - - /** Invalidate every owner's in-flight returns, including owners not yet seen. */ - private bumpAll(): void { - for (const owner of new Set([...this.entries.values()].map((entry) => entry.owner))) this.bump(owner); - for (const owner of [...this.generations.keys()]) this.bump(owner); } private dropIfExpired(): void { if (this.builtAt > 0 && Date.now() - this.builtAt > PLAN_TTL_MS) { - // Bump before clearing, while the owners are still known: a restore that - // took entries just before the expiry must not hand them back afterwards - // and give an expired plan another full day of life. - this.bumpAll(); + // A restore that took entries just before the expiry must not hand them + // back afterwards and give an expired plan another full day of life. + this.parked.clear(); this.entries.clear(); this.builtAt = 0; } diff --git a/src/web/routes/reboot-restore-routes.ts b/src/web/routes/reboot-restore-routes.ts index da742cc4..3516ff19 100644 --- a/src/web/routes/reboot-restore-routes.ts +++ b/src/web/routes/reboot-restore-routes.ts @@ -92,8 +92,7 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes if (!rebootRestoreRegistry.beginSpending(owner)) { return reply.code(409).send(createErrorResponse(ApiErrorCode.CONFLICT, 'A reboot restore is already running')); } - const taken = rebootRestoreRegistry.take(canAccess, body.sessionIds); - const generations = rebootRestoreRegistry.snapshotGenerations(taken); + const taken = rebootRestoreRegistry.take(canAccess, body.sessionIds, owner); // Entries nothing built a pane for, returned to the plan on every exit path // including a throw. Without this a failure between here and the loop would // spend the offer and rebuild nothing, and the plan cannot be rebuilt. @@ -252,9 +251,9 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes } finally { // Anything that never became a pane goes back on offer, including after a // throw, so a transient failure costs a retry rather than the whole plan. - // Passing the generations makes a Dismiss that landed mid-restore win, for - // the owners it actually covered. - rebootRestoreRegistry.restore([...unspent], generations); + // Ends the flight: entries still parked for it come back if they are in + // `unspent`, and a Dismiss that unparked them meanwhile wins. + rebootRestoreRegistry.releaseFlight(owner, [...unspent]); rebootRestoreRegistry.endSpending(owner); } }); diff --git a/src/web/server.ts b/src/web/server.ts index 095b094d..2358b2ff 100644 --- a/src/web/server.ts +++ b/src/web/server.ts @@ -3010,26 +3010,42 @@ export class WebServer extends EventEmitter { if (!session) return; this.sessions.delete(sessionId); - // --- the inverse of setupSessionListeners(), in its order --- - const summaryTracker = this.runSummaryTrackers.get(sessionId); - if (summaryTracker) { - summaryTracker.stop(); - this.runSummaryTrackers.delete(sessionId); - } - // An fs.watch on the workspace (or on @fix_plan.md) that nothing else closes. - session.ralphTracker.stopWatchingFixPlan(); - // An FSWatcher on the workspace, likewise. - imageWatcher.unwatchSession(sessionId); + // --- the inverse of setupSessionListeners(), in reverse order --- + // Listeners first: while they are attached, one of them can still reach a + // tracker this is about to stop. const listeners = this.sessionListenerRefs.get(sessionId); if (listeners) { detachSessionListeners(session, listeners); this.sessionListenerRefs.delete(sessionId); } + // An FSWatcher on the workspace that nothing else closes. + imageWatcher.unwatchSession(sessionId); + // An fs.watch on the workspace (or on @fix_plan.md), likewise. + session.ralphTracker.stopWatchingFixPlan(); + const summaryTracker = this.runSummaryTrackers.get(sessionId); + if (summaryTracker) { + summaryTracker.stop(); + this.runSummaryTrackers.delete(sessionId); + } + + // --- what anything else may have attached to this id in the meantime --- + // A rebuild can fail AFTER startInteractive() resolved, and a restored + // workspace still carries Codeman's hooks, so the CLI can post a hook event + // within milliseconds. Each of these outlives the listeners and would + // otherwise meet the retry, which reuses the same session id by design. + this.stopTranscriptWatcher(sessionId); + attachmentRegistry.clearSession(sessionId); + sessionWaits.notifySignal(sessionId, 'exit'); + sessionWaits.cancelAll(sessionId); + approvalInbox.resolveForSession(sessionId, 'session_ended'); // --- the inverse of the construction itself --- this.sse.cleanupSessionBatches(sessionId); this.persistDeb.cancelKey(sessionId); fileStreamManager.closeSessionStreams(sessionId); + // `lastRecordedTokens` is deliberately NOT deleted: the `after-spawn` phase + // seeds it as the daily-usage baseline for these restored totals, and the + // retry reuses the id, so dropping it would count them as new usage. // The per-session custom-model config dir carries the endpoint's API key, and // `before-spawn` may already have written it. Nothing else would ever remove // it: the stale sweep only touches state.json. A retry rewrites it. @@ -3039,6 +3055,9 @@ export class WebServer extends EventEmitter { await session.stop(true); } catch (err) { console.warn(`[Server] stopping a partially built session failed: ${getErrorMessage(err)}`); + // `stop()` kills the mux session in its last block, after destroying its + // trackers, so a throw on the way there leaves the pane running. + await this.mux.killSession(sessionId).catch(() => {}); } try { await this.tabLayouts.sessionsRemoved([{ id: sessionId, owner: session.owner }]); diff --git a/test/discard-partially-built-session.test.ts b/test/discard-partially-built-session.test.ts index 7232e969..012009f4 100644 --- a/test/discard-partially-built-session.test.ts +++ b/test/discard-partially-built-session.test.ts @@ -18,10 +18,11 @@ * leaves that entry makes the next attempt wire nothing at all, and the user * gets a tab that never shows output. */ -import { mkdirSync, rmSync } from 'node:fs'; +import { mkdirSync } from 'node:fs'; import { homedir } from 'node:os'; import { join } from 'node:path'; -import { afterEach, beforeEach, describe, expect, it } from 'vitest'; +import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest'; +import { safeRmHomeTree } from './mocks/test-helpers.js'; import { WebServer } from '../src/web/server.js'; import { Session } from '../src/session.js'; @@ -55,9 +56,11 @@ function buildSession(): Session { }); } -beforeEach(() => { +beforeAll(() => { mkdirSync(WORKSPACE, { recursive: true }); - // Test mode: no port is opened and no CLI is launched. + // Test mode: no port is opened and no CLI is launched. One server for the file, + // stopped at the end: the constructor registers handlers on the module-level + // image, subagent, team and workflow watchers, and only stop() removes them. server = new WebServer(0, false, true); internals = server as unknown as ServerInternals; mux = new TmuxManager(); @@ -65,7 +68,11 @@ beforeEach(() => { afterEach(async () => { await internals.discardPartiallyBuiltSession(SESSION_ID).catch(() => {}); - rmSync(WORKSPACE, { recursive: true, force: true }); +}); + +afterAll(async () => { + await server.stop().catch(() => {}); + safeRmHomeTree(WORKSPACE); }); describe('discarding a session whose pane never started', () => { @@ -85,25 +92,36 @@ describe('discarding a session whose pane never started', () => { await internals.setupSessionListeners(first); expect(internals.sessionListenerRefs.has(SESSION_ID)).toBe(true); + const firstRefs = internals.sessionListenerRefs.get(SESSION_ID); await internals.discardPartiallyBuiltSession(SESSION_ID); expect(internals.sessionListenerRefs.has(SESSION_ID)).toBe(false); // The retry reuses the id by design. `setupSessionListeners()` returns early - // while the refs are still there, so a session built now would run blind: - // no terminal output, no status updates, no exit broadcast. + // while the refs are still there, so a session built now would run blind: no + // terminal output, no status updates, no exit broadcast. Asserting a DIFFERENT + // refs object is what distinguishes wiring the retry from finding the corpse + // of the first attempt still in place. const retry = buildSession(); await internals.registerSessionWithLayout(retry); await internals.setupSessionListeners(retry); - expect(internals.sessionListenerRefs.has(SESSION_ID)).toBe(true); + const retryRefs = internals.sessionListenerRefs.get(SESSION_ID); + expect(retryRefs).toBeDefined(); + expect(retryRefs).not.toBe(firstRefs); }); it('stops the run-summary tracker, whose interval would otherwise keep firing', async () => { const session = buildSession(); await internals.registerSessionWithLayout(session); await internals.setupSessionListeners(session); - expect(internals.runSummaryTrackers.has(SESSION_ID)).toBe(true); + const tracker = internals.runSummaryTrackers.get(SESSION_ID) as { stop: () => void }; + expect(tracker).toBeDefined(); + // Dropping the map entry is not enough: the tracker arms a setInterval in its + // constructor, and only stop() clears it, so a discard that merely forgot the + // entry would leave the timer running for the life of the process. + const stopped = vi.spyOn(tracker, 'stop'); await internals.discardPartiallyBuiltSession(SESSION_ID); + expect(stopped).toHaveBeenCalled(); expect(internals.runSummaryTrackers.has(SESSION_ID)).toBe(false); }); diff --git a/test/reboot-restore.test.ts b/test/reboot-restore.test.ts index 76506358..aaf9a83e 100644 --- a/test/reboot-restore.test.ts +++ b/test/reboot-restore.test.ts @@ -198,15 +198,15 @@ describe('the plan the banner spends', () => { it('hands an entry to the first caller and nothing to the second', () => { const registry = new RebootRestoreRegistry(); registry.set([entryFor('a'), entryFor('b')]); - expect(registry.take(all).map((e) => e.sessionId)).toEqual(['a', 'b']); + expect(registry.take(all, undefined, undefined).map((e) => e.sessionId)).toEqual(['a', 'b']); // The double-click: two panes on one conversation is what this prevents. - expect(registry.take(all)).toEqual([]); + expect(registry.take(all, undefined, undefined)).toEqual([]); }); it('spends only the ids a caller asked for', () => { const registry = new RebootRestoreRegistry(); registry.set([entryFor('a'), entryFor('b')]); - expect(registry.take(all, ['b']).map((e) => e.sessionId)).toEqual(['b']); + expect(registry.take(all, ['b'], undefined).map((e) => e.sessionId)).toEqual(['b']); expect(registry.list(all).map((e) => e.sessionId)).toEqual(['a']); }); @@ -215,7 +215,7 @@ describe('the plan the banner spends', () => { registry.set([entryFor('mine', 'alice'), entryFor('theirs', 'bob')]); const asAlice = (owner: string | undefined) => owner === 'alice'; expect(registry.list(asAlice).map((e) => e.sessionId)).toEqual(['mine']); - expect(registry.take(asAlice).map((e) => e.sessionId)).toEqual(['mine']); + expect(registry.take(asAlice, undefined, 'alice').map((e) => e.sessionId)).toEqual(['mine']); // Bob's entry is still on offer for Bob. expect(registry.list(() => true).map((e) => e.sessionId)).toEqual(['theirs']); }); @@ -223,8 +223,8 @@ describe('the plan the banner spends', () => { it('puts back an entry that no pane was created for', () => { const registry = new RebootRestoreRegistry(); registry.set([entryFor('a')]); - const taken = registry.take(all); - registry.restore(taken); + const taken = registry.take(all, undefined, undefined); + registry.releaseFlight(undefined, taken); expect(registry.list(all).map((e) => e.sessionId)).toEqual(['a']); }); diff --git a/test/routes/reboot-restore-rebuild-failure.test.ts b/test/routes/reboot-restore-rebuild-failure.test.ts index 9ee93fb4..7a69e30d 100644 --- a/test/routes/reboot-restore-rebuild-failure.test.ts +++ b/test/routes/reboot-restore-rebuild-failure.test.ts @@ -183,12 +183,23 @@ describe('a rebuild that succeeds', () => { callOrder.push(`reapply:${phase}`); } ); + (ctx.setupSessionListeners as ReturnType<typeof vi.fn>).mockImplementation(async () => { + callOrder.push('setupSessionListeners'); + }); const app = await createHarness(ctx); await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} }); - // The custom-model environment has to reach the process; the token totals - // must not land on a session whose pane never started. - expect(callOrder).toEqual(['reapply:before-spawn', 'startInteractive', 'reapply:after-spawn']); + // `setupSessionListeners()` READS the image-watcher flag that `before-spawn` + // restores, so the phase has to precede it or the session comes back + // reporting the watcher as on with nothing watching. The custom-model + // environment has to reach the process, and the token totals must not land + // on a session whose pane never started. + expect(callOrder).toEqual([ + 'reapply:before-spawn', + 'setupSessionListeners', + 'startInteractive', + 'reapply:after-spawn', + ]); await app.close(); }); @@ -282,6 +293,32 @@ describe('a failure before any entry is considered', () => { }); describe('a dismiss that lands while a restore is running', () => { + it('wins when an admin is restoring the entries and their owner dismisses', async () => { + const theirs = offerEntry('theirs', 'bob'); + rebootRestoreRegistry.set([theirs]); + // An admin may spend another user's entries, so the caller doing the restore + // and the owner of what is being restored are different people. + const taken = rebootRestoreRegistry.take(() => true, undefined, 'admin'); + expect(taken.map((e) => e.sessionId)).toEqual(['theirs']); + + // Bob dismisses his own banner. Nothing of his is in the plan any more, and + // the restore is running under a different name than his. + rebootRestoreRegistry.clear((owner) => owner === 'bob'); + rebootRestoreRegistry.releaseFlight('admin', taken); + + expect(rebootRestoreRegistry.list(() => true)).toEqual([]); + }); + + it('wins when an admin dismisses everything mid-restore', async () => { + rebootRestoreRegistry.set([offerEntry('theirs', 'bob')]); + const taken = rebootRestoreRegistry.take(() => true, undefined, 'admin'); + + rebootRestoreRegistry.clear(() => true); + rebootRestoreRegistry.releaseFlight('admin', taken); + + expect(rebootRestoreRegistry.list(() => true)).toEqual([]); + }); + it('wins, rather than being undone when the route hands its entries back', async () => { rebootRestoreRegistry.set([offerEntry('a')]); const ctx = createMockRouteContext({ workspaceHooksEnabled: false }); @@ -304,15 +341,13 @@ describe('a dismiss that lands while a restore is running', () => { it('reaches an in-flight restore the dismisser can see, even once its entries are taken', async () => { const mine = offerEntry('mine', 'alice'); rebootRestoreRegistry.set([mine]); - expect(rebootRestoreRegistry.beginSpending('alice')).toBe(true); - const generations = rebootRestoreRegistry.snapshotGenerations([mine]); - const taken = rebootRestoreRegistry.take((owner) => owner === 'alice'); + const taken = rebootRestoreRegistry.take((owner) => owner === 'alice', undefined, 'alice'); + expect(taken).toHaveLength(1); - // The plan is empty now, so a dismiss has nothing of Alice's to remove; the - // invalidation has to come from her claimed flight. + // The plan is empty now, so the dismiss has nothing of Alice's left in the + // plan; it has to reach the entry the restore is holding. rebootRestoreRegistry.clear((owner) => owner === 'alice'); - rebootRestoreRegistry.restore(taken, generations); - rebootRestoreRegistry.endSpending('alice'); + rebootRestoreRegistry.releaseFlight('alice', taken); expect(rebootRestoreRegistry.list(() => true)).toEqual([]); }); @@ -322,13 +357,8 @@ describe('a dismiss that lands while a restore is running', () => { const theirs = offerEntry('theirs', 'bob'); rebootRestoreRegistry.set([mine, theirs]); - // Bob is mid-restore, holding his own entry. The claimed flight is what makes - // this the interesting case: a dismiss can no longer see Bob's entries in the - // plan, so the invalidation has to come from the in-flight set, filtered by - // what the dismissing user may access. - expect(rebootRestoreRegistry.beginSpending('bob')).toBe(true); - const bobsGenerations = rebootRestoreRegistry.snapshotGenerations([theirs]); - const bobsTaken = rebootRestoreRegistry.take((owner) => owner === 'bob'); + // Bob is mid-restore, holding his own entry. + const bobsTaken = rebootRestoreRegistry.take((owner) => owner === 'bob', undefined, 'bob'); expect(bobsTaken.map((e) => e.sessionId)).toEqual(['theirs']); // Alice dismisses her own banner meanwhile. @@ -336,8 +366,7 @@ describe('a dismiss that lands while a restore is running', () => { // Bob's restore finishes and hands his entry back. Alice's dismiss covered // her entries, not his, so his offer survives. - rebootRestoreRegistry.restore(bobsTaken, bobsGenerations); - rebootRestoreRegistry.endSpending('bob'); + rebootRestoreRegistry.releaseFlight('bob', bobsTaken); expect(rebootRestoreRegistry.list(() => true).map((e) => e.sessionId)).toEqual(['theirs']); }); }); From 18ab2ab5950c941bb4f9269c71f3ab7f76811837 Mon Sep 17 00:00:00 2001 From: Michael Grundberg <michael.grundberg@irisity.com> Date: Thu, 17 Sep 2026 08:26:29 +0200 Subject: [PATCH 15/28] docs(sessions): correct what a failed rebuild is actually likely to be MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ran the feature against a real server for the first time, on an isolated instance, and two claims in the code turned out to be wrong. A rebuild that fails after the session is registered was documented as commonly caused by a CLI binary missing from a freshly booted machine's PATH. It is not: the resolver finds its binary by absolute path, so PATH never enters into it, and a server started without claude on PATH restored every session normally. Nor does an un-enterable workspace fail — tmux falls back to another directory and the pane comes up there. Neither obvious cause throws, so the discard path is defended rather than expected, and the comments now say that instead of naming a cause that cannot happen. The four review rounds that shaped this path all reasoned about a trigger none of them could test. The path itself is still worth having, since a mux failure would reach it, but its comments should not claim a likelihood the machine disagrees with. Refs #411 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- src/web/routes/reboot-restore-routes.ts | 10 ++++++++-- test/routes/reboot-restore-rebuild-failure.test.ts | 13 +++++++++---- 2 files changed, 17 insertions(+), 6 deletions(-) diff --git a/src/web/routes/reboot-restore-routes.ts b/src/web/routes/reboot-restore-routes.ts index 3516ff19..3ff27b4f 100644 --- a/src/web/routes/reboot-restore-routes.ts +++ b/src/web/routes/reboot-restore-routes.ts @@ -220,8 +220,14 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes // One entry that will not start must not stop the rest of the pass, and // must not leave a registered session with no pane behind it: by this // point the session is in `ctx.sessions`, holds a tab-layout slot and has - // listeners, and the commonest cause is a CLI binary that is not on the - // PATH of a freshly booted machine. + // listeners. + // + // Reaching this is rarer than it looks, measured against a real server: + // the CLI resolver finds its binary by absolute path rather than through + // PATH, and tmux falls back to another directory rather than failing when + // it cannot enter the workspace, so neither of the two obvious "freshly + // booted machine" failures throws. What is left is the mux layer itself + // failing, which is why this path is defended rather than expected. console.error(`[reboot-restore] failed to rebuild ${entry.sessionId}:`, err); // Not cleanupSession(): that is the user-initiated delete, and it would // count this session's historical tokens into the lifetime totals, demote diff --git a/test/routes/reboot-restore-rebuild-failure.test.ts b/test/routes/reboot-restore-rebuild-failure.test.ts index 7a69e30d..98cf7a17 100644 --- a/test/routes/reboot-restore-rebuild-failure.test.ts +++ b/test/routes/reboot-restore-rebuild-failure.test.ts @@ -4,10 +4,15 @@ * The other route test file deliberately uses workspaces that do not exist, so it * never reaches `new Session()`. This one mocks the `Session` module so the route * runs its whole construction path — `addSession`, `setupSessionListeners`, - * `reapplyPersistedSessionState`, `startInteractive` — and then throws where a - * real one would when the CLI binary is missing from a freshly booted machine's - * PATH. Without the mock there is no way to exercise that path, which is how the - * original version of this route shipped a session leak the tests could not see. + * `reapplyPersistedSessionState`, `startInteractive` — and then throws. + * + * The mock is the only way in. Driven against a real server, `startInteractive()` + * does not throw for either obvious cause: the CLI resolver finds its binary by + * absolute path rather than through PATH, and tmux falls back to another + * directory rather than failing when it cannot enter the workspace. A mux-layer + * failure is what is left, and it cannot be provoked from a test. Without the + * mock this path would go unexercised, which is how the original version of this + * route shipped a session leak the tests could not see. * * It also covers the session caps, because those too are only reachable once the * route is actually willing to build something. From 5f55f9cb65b39716461dd26854dacb91089c2ade Mon Sep 17 00:00:00 2001 From: Michael Grundberg <michael.grundberg@irisity.com> Date: Thu, 17 Sep 2026 14:11:58 +0200 Subject: [PATCH 16/28] fix(sessions): never let the reboot-restore plan fail recovery The plan build runs inside the try that decides whether restoreMuxSessions() succeeded, so a throw would be caught there, report restoration as failed, and block the stale cleanup and layout reconciliation that follow. An optional convenience would then break the recovery it exists to help. It is guarded on its own now: the correct way for this to fail is an offer nobody gets. Refs #411 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- src/web/server.ts | 13 ++++++++++++- 1 file changed, 12 insertions(+), 1 deletion(-) diff --git a/src/web/server.ts b/src/web/server.ts index 2358b2ff..e94b8cdf 100644 --- a/src/web/server.ts +++ b/src/web/server.ts @@ -3081,7 +3081,18 @@ export class WebServer extends EventEmitter { // Build the reboot-restore offer HERE: `dead` is only known after // reconciliation, and the records it reads are pruned by // `cleanupStaleSessions()` as soon as `finalizeRestoredState()` runs. - this.planRebootRestoreOffer(dead, alive.length); + // + // Guarded on its own, because this runs inside the try that decides whether + // RECOVERY succeeded. A throw here would otherwise be caught below, report + // restoration as failed, and block the stale cleanup and layout + // reconciliation that follow — turning an optional convenience into a + // failure of the thing it is supposed to help. An offer nobody gets is the + // correct way for this to fail. + try { + this.planRebootRestoreOffer(dead, alive.length); + } catch (err) { + console.error('[Server] Building the reboot-restore offer failed; continuing recovery:', err); + } if (alive.length > 0 || discovered.length > 0) { console.log(`[Server] Found ${alive.length + discovered.length} alive mux session(s) from previous run`); From 5108a24bf0f63eff8d2dd92e661f668a148219c7 Mon Sep 17 00:00:00 2001 From: Michael Grundberg <michael.grundberg@irisity.com> Date: Thu, 17 Sep 2026 18:46:15 +0200 Subject: [PATCH 17/28] fix(sessions): never restore a session whose agent was already exited MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found by a real reboot, which is the first thing to catch it. Typing `/exit` ends the CLI process and leaves the session record behind, and the process-exit handler persists `pid: null` with `status: 'idle'` before anything else runs. By status alone that is indistinguishable from a session sitting idle when the power went, so the boot pass offered those sessions back and a click spawned the agents the user had deliberately closed — the exact case the eligibility rule exists to exclude. The absent pid is what tells the two apart, and the plan step now refuses a record without one, under its own `not-running` reason so the boot log says why. On a healthy board every running session carries a pid; a record with none describes an agent that is already gone. Deliberately the conservative direction. A session that somehow persisted no pid while genuinely running is not offered, and its conversation stays reachable from the Resume list, which is where every session would be without this feature. The opposite error spawns processes nobody asked for. Ark0N/Codeman#446 covers the dead panes those exits leave behind, but this does not wait on it: the rule belongs here whether or not the record's shape changes later. Refs #411 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- src/reboot-restore.ts | 25 ++++++++++++++++++++++++- test/reboot-restore.test.ts | 27 +++++++++++++++++++++++++++ 2 files changed, 51 insertions(+), 1 deletion(-) diff --git a/src/reboot-restore.ts b/src/reboot-restore.ts index e226138c..d295a34d 100644 --- a/src/reboot-restore.ts +++ b/src/reboot-restore.ts @@ -19,6 +19,13 @@ * touching its status, so a pinned session a reboot killed still reads `idle` or * `busy` and stays eligible. * + * Ending the AGENT rather than the session leaves a third shape, and it is the one + * a real reboot caught this module getting wrong. `/exit` ends the CLI process + * while the session record survives, and the process-exit handler persists + * `pid: null` with `status: 'idle'` — indistinguishable by status from a session + * that was merely idle when the power went. The absent pid is what tells them + * apart, so a record without one is refused. + * * @dependencies types (SessionState), config/cli-registry * @consumedby web/server (plan build at boot), web/routes/reboot-restore-routes * @@ -89,7 +96,7 @@ export function resolveResumeConversationId(state: SessionState): string { /** * Why one session was passed over. Reported for logging and shown to the user. * - * The first six are decided before anything is built. `capacity-reached` and + * The first seven are decided before anything is built. `capacity-reached` and * `rebuild-failed` can only happen once a click is spending the plan, and they * are the two the banner must not confuse with a missing workspace: one means * "try again after closing something", the other means the CLI would not start. @@ -99,6 +106,7 @@ export interface RebootRestoreRejection { reason: | 'no-persisted-record' | 'intentionally-ended' + | 'not-running' | 'respawn-blocked' | 'remote-or-docker' | 'unsupported-mode' @@ -163,6 +171,21 @@ export function planRebootRestore( skipped.push({ sessionId, reason: 'intentionally-ended' }); continue; } + if (state.pid === null || state.pid === undefined) { + // The agent had already exited when the machine went down: `/exit` ends the + // process, and its exit handler persists `pid: null` with `status: 'idle'` + // before anything else can. Status alone cannot tell that apart from a + // session that was simply sitting idle when the power went, so without this + // a reboot restore spawns the agents the user deliberately closed — the + // exact case the eligibility rule exists to exclude. + // + // A heuristic, and deliberately the conservative one. A session that somehow + // persisted no pid while genuinely running is not offered, and its + // conversation stays reachable from the Resume list, which is where every + // session would be without this feature. + skipped.push({ sessionId, reason: 'not-running' }); + continue; + } if (state.respawnBlocked === true) { // The crash-loop breaker tripped on this pane. Re-creating it restarts the loop. skipped.push({ sessionId, reason: 'respawn-blocked' }); diff --git a/test/reboot-restore.test.ts b/test/reboot-restore.test.ts index aaf9a83e..5ed989fd 100644 --- a/test/reboot-restore.test.ts +++ b/test/reboot-restore.test.ts @@ -39,6 +39,7 @@ const NOW = 1_760_000_000_000; function persistedSession(overrides: Partial<SessionState> & { id: string }): SessionState { return { + // A live agent's record carries its process id; `/exit` persists null instead. pid: 99999, status: 'idle', workingDir: '/tmp/spike', @@ -137,6 +138,32 @@ describe('which dead sessions may be rebuilt', () => { }); }); +describe('a session whose agent had already exited', () => { + it('is refused, because `/exit` leaves the record reading idle with no pid', () => { + // What the process-exit handler persists: the CLI is gone, the record is not, + // and its status is indistinguishable from a session that was merely idle. + const persisted = { exited: persistedSession({ id: 'exited', status: 'idle', pid: null }) }; + const plan = planRebootRestore(['exited'], persisted, () => true); + expect(plan.restore).toEqual([]); + expect(plan.skipped).toEqual([{ sessionId: 'exited', reason: 'not-running' }]); + }); + + it('still restores the session beside it that was running when the power went', () => { + const persisted = { + exited: persistedSession({ id: 'exited', pid: null }), + running: persistedSession({ id: 'running', pid: 4242 }), + }; + const plan = planRebootRestore(['exited', 'running'], persisted, () => true); + expect(plan.restore.map((entry) => entry.sessionId)).toEqual(['running']); + expect(plan.skipped.map((s) => s.reason)).toEqual(['not-running']); + }); + + it('refuses a record with no pid field at all', () => { + const persisted = { odd: persistedSession({ id: 'odd', pid: undefined as unknown as null }) }; + expect(planRebootRestore(['odd'], persisted, () => true).skipped[0].reason).toBe('not-running'); + }); +}); + describe('a workspace that is no longer on disk', () => { it('is kept out of the offer, so a click cannot scaffold a deleted repo', () => { const persisted = { gone: persistedSession({ id: 'gone', workingDir: '/tmp/deleted-repo' }) }; From 62ceb4e87b35904b28a039c8102f3dde938b725f Mon Sep 17 00:00:00 2001 From: Michael Grundberg <michael.grundberg@irisity.com> Date: Thu, 17 Sep 2026 19:49:06 +0200 Subject: [PATCH 18/28] fix(sessions): correct what the missing-pid rule actually recognises MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A second real reboot disproved the mechanism the previous commit was built on. Typing `/exit` does not persist `pid: null`, and the session was restored anyway. The pid a session record carries is its `tmux attach-session` process, not the agent. `/exit` ends the CLI inside the pane, `remain-on-exit` keeps the pane, and the attach process stays alive throughout — so Codeman's PTY never exits, no exit handler runs, and the record keeps both its pid and `status: 'idle'`. The lifecycle log for the session that came back shows created, started, stale_cleaned and recovered, with no exit event at all, which is the proof: Codeman never learned the agent was gone. So nothing durable distinguishes an exited agent from a session that was idle when the power went, and this pass restores both. Ark0N/Codeman#446 is about making Codeman notice the dead pane; contrary to what the previous commit's message claimed, this genuinely does wait on that. Until a record can say the agent is gone, the user dismisses or closes those sessions. The rule itself is kept, because a record with no attach process does describe a session that never started or whose pane died outright, and refusing it is right. Only its documentation was wrong. The module header, the branch comment and the test names now say what it recognises instead of claiming the case it cannot see. Refs #411 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- src/reboot-restore.ts | 40 ++++++++++++++++++++++--------------- test/reboot-restore.test.ts | 11 +++++----- 2 files changed, 30 insertions(+), 21 deletions(-) diff --git a/src/reboot-restore.ts b/src/reboot-restore.ts index d295a34d..85760e8d 100644 --- a/src/reboot-restore.ts +++ b/src/reboot-restore.ts @@ -19,12 +19,19 @@ * touching its status, so a pinned session a reboot killed still reads `idle` or * `busy` and stays eligible. * - * Ending the AGENT rather than the session leaves a third shape, and it is the one - * a real reboot caught this module getting wrong. `/exit` ends the CLI process - * while the session record survives, and the process-exit handler persists - * `pid: null` with `status: 'idle'` — indistinguishable by status from a session - * that was merely idle when the power went. The absent pid is what tells them - * apart, so a record without one is refused. + * ⚠️ Ending the AGENT rather than the session is a shape this module CANNOT + * recognise today, and a reboot restores it. `/exit` ends the CLI inside the + * pane, `remain-on-exit` keeps the pane, and the PTY Codeman owns is the + * `tmux attach-session` process, which stays alive throughout — so no exit + * handler runs, no lifecycle `exit` is logged, and the record keeps both its pid + * and `status: 'idle'`. Nothing durable distinguishes it from a session that was + * simply idle when the power went. Ark0N/Codeman#446 covers making Codeman + * notice the dead pane; until a record can say the agent is gone, this pass will + * offer those sessions back, and the user dismisses or closes them. + * + * The `pid` check below is therefore NOT that rule. It refuses a record whose + * attach process was already gone, which is a session that never started or + * whose pane died outright. * * @dependencies types (SessionState), config/cli-registry * @consumedby web/server (plan build at boot), web/routes/reboot-restore-routes @@ -172,17 +179,18 @@ export function planRebootRestore( continue; } if (state.pid === null || state.pid === undefined) { - // The agent had already exited when the machine went down: `/exit` ends the - // process, and its exit handler persists `pid: null` with `status: 'idle'` - // before anything else can. Status alone cannot tell that apart from a - // session that was simply sitting idle when the power went, so without this - // a reboot restore spawns the agents the user deliberately closed — the - // exact case the eligibility rule exists to exclude. + // No attach process when the record was last written: the session never + // started, or its pane died outright rather than its agent exiting inside a + // surviving pane. Either way there was nothing running to bring back. // - // A heuristic, and deliberately the conservative one. A session that somehow - // persisted no pid while genuinely running is not offered, and its - // conversation stays reachable from the Resume list, which is where every - // session would be without this feature. + // ⚠️ This does NOT catch a session the user ended with `/exit`. See the + // module header: that leaves the pid in place, because the pid is the tmux + // attach process and `remain-on-exit` keeps it alive. + // + // Conservative on purpose. A session that somehow persisted no pid while + // genuinely running is not offered, and its conversation stays reachable + // from the Resume list, which is where every session would be without this + // feature. skipped.push({ sessionId, reason: 'not-running' }); continue; } diff --git a/test/reboot-restore.test.ts b/test/reboot-restore.test.ts index 5ed989fd..67bc34e9 100644 --- a/test/reboot-restore.test.ts +++ b/test/reboot-restore.test.ts @@ -138,17 +138,18 @@ describe('which dead sessions may be rebuilt', () => { }); }); -describe('a session whose agent had already exited', () => { - it('is refused, because `/exit` leaves the record reading idle with no pid', () => { - // What the process-exit handler persists: the CLI is gone, the record is not, - // and its status is indistinguishable from a session that was merely idle. +describe('a session with no attach process in its record', () => { + it('is refused, because there was nothing running to bring back', () => { + // A session that never started, or whose pane died outright. NOT a session + // the user ended with `/exit`: that keeps its pid, because the pid is the + // tmux attach process and `remain-on-exit` keeps the pane alive. const persisted = { exited: persistedSession({ id: 'exited', status: 'idle', pid: null }) }; const plan = planRebootRestore(['exited'], persisted, () => true); expect(plan.restore).toEqual([]); expect(plan.skipped).toEqual([{ sessionId: 'exited', reason: 'not-running' }]); }); - it('still restores the session beside it that was running when the power went', () => { + it('still restores the session beside it that was attached when the power went', () => { const persisted = { exited: persistedSession({ id: 'exited', pid: null }), running: persistedSession({ id: 'running', pid: 4242 }), From 1f61d2129874d0ccbaafb7528963e156ce22c9b4 Mon Sep 17 00:00:00 2001 From: Codeman maintainer <noreply@anthropic.com> Date: Fri, 18 Sep 2026 13:41:02 +0200 Subject: [PATCH 19/28] docs: correct six stale counts and claims in CLAUDE.md Each of these was measurable and wrong: the CI note listed 5 excluded Playwright tests where config/test-suites.ts has 9, never mentioned the packages/xterm-zerolag-input run that follows the gate, and never mentioned wiki-sync.yml at all; the format glob note omitted that lint covers only src/**/*.ts; app.js is ~6.9K lines, not ~6.7K, and voice-pcm-worklet.js is fetched from JS rather than sitting in the load order; src/config/ holds 23 files plus the cli-registry/ subdir, not 21, and nothing said that the repo-root config/ is a different directory; the route count is ~232 with cases at 34, not ~228 with cases at 30. Also adds the pointer to docs/wiki/ as the user-facing manual, which the header describes every other doc surface but not that one. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- CLAUDE.md | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 4ef118e5..7278266d 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -2,7 +2,7 @@ This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository. -> Deep implementation detail lives in [`docs/architecture-invariants.md`](docs/architecture-invariants.md). This file holds the rules that prevent mistakes; that file holds the mechanisms, file inventories, and the history behind each rule. Pointers below are written as `→ architecture-invariants#anchor`. When the goal is raw throughput, [`docs/SPEEDRUN.md`](docs/SPEEDRUN.md) is the fast-execution protocol (it removes ceremony, never the safety rules here). +> Deep implementation detail lives in [`docs/architecture-invariants.md`](docs/architecture-invariants.md). This file holds the rules that prevent mistakes; that file holds the mechanisms, file inventories, and the history behind each rule. Pointers below are written as `→ architecture-invariants#anchor`. When the goal is raw throughput, [`docs/SPEEDRUN.md`](docs/SPEEDRUN.md) is the fast-execution protocol (it removes ceremony, never the safety rules here). The user-facing manual is `docs/wiki/` (mirrored to the GitHub wiki by CI; see the CI note under Additional Commands), and `AGENTS.md` deliberately just points here. > > **This file is in `.prettierignore` on purpose.** Prettier's markdown printer escapes underscores inside the glob-heavy paths used throughout (`agent-*.jsonl` became `agent-\_.jsonl`, collapsing backtick spans and corrupting a whole paragraph). Do not remove the ignore entry, and do not run `prettier --write` on it. > @@ -120,11 +120,11 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph | Dependency doctor | `codeman doctor` (alias `check-deps`; `--json`, `--category core\|office\|other`). Probes Node/Claude CLI/tmux/LibreOffice/MS Office against `config/dependency-registry.ts`; engine is pure given an injectable `ProbeHost` | | Multi-user accounts | `codeman users add <name>` / `passwd <name>` / `list` / `rm <name>` (writes `~/.codeman/users.json`, mode 0600; see Multi-user mode) | -**CI**: `.github/workflows/ci.yml` (push to master/main + PRs, Node 22) runs two jobs: **(1)** `check:lockfile`, `typecheck`, `lint`, `check:frontend-syntax`, `format:check`, then a **server boot smoke test** (`tsx src/index.ts web --port 3151` must answer `/api/status` within 30s); **(2)** the **unit/integration test suite** via `npm run test:ci` (`config/vitest.ci.config.ts` — excludes the browser-driven `test/mobile/**` suite, `perf-*` benchmarks, and 5 Playwright tests; globs live in `config/test-suites.ts`). `npm test` runs this same config, so local green == CI green. Tests are tmux-safe in CI: `TmuxManager` no-ops all shell commands under `VITEST` (see Testing). +**CI**: `.github/workflows/ci.yml` (push to master/main + PRs, Node 22) runs two jobs: **(1)** `check:lockfile`, `typecheck`, `lint`, `check:frontend-syntax`, `format:check`, then a **server boot smoke test** (`tsx src/index.ts web --port 3151` must answer `/api/status` within 30s); **(2)** the **unit/integration test suite** via `npm run test:ci` (`config/vitest.ci.config.ts` — excludes the browser-driven `test/mobile/**` suite, `perf-*` benchmarks, and 9 Playwright tests; globs live in `config/test-suites.ts`), followed by the **`packages/xterm-zerolag-input` package tests** (a bare `npx vitest run` in that directory; its vitest is hoisted by the root `npm ci`, so no separate install, and `npm test` at the root does NOT run them). `npm test` runs this same config, so local green == CI green. Tests are tmux-safe in CI: `TmuxManager` no-ops all shell commands under `VITEST` (see Testing). A third workflow, `wiki-sync.yml`, fires only on master pushes touching `docs/wiki/**` and mirrors that directory to the GitHub wiki (browser edits to the wiki are overwritten by the next sync, so fix pages via `docs/wiki/`). **Code style**: Prettier (`singleQuote: true`, `printWidth: 120`, `trailingComma: "es5"`) — config lives in the **`"prettier"` key of `package.json`**, not a `.prettierrc` (keeps the repo root short; editors read it natively). `.prettierignore` stays at the root because Prettier resolves it relative to cwd. ESLint flat config (`config/eslint.config.js`) allows `no-console`, warns on `@typescript-eslint/no-explicit-any`. Ignores: `app.js`, `scripts/**/*.mjs`, `src/web/public/vendor/**`, `scripts/remotion/**`. -**Prettier scope is deliberately narrow.** `npm run format` globs only `src/**/*.ts` and `src/web/public/**`, and `.prettierignore` then exempts most of `src/web/public/*.js` (app.js, styles.css, **mobile.css**, index.html, upload.html, and 15 hand-formatted modules) plus `CLAUDE.md`. Those files are hand-formatted by design; `npm run check:public-assets` and `check:frontend-syntax` are what guard them (NUL bytes + JS syntax), not Prettier. Do not "fix" a file by adding it back to Prettier's scope. +**Prettier scope is deliberately narrow.** `npm run format` globs only `src/**/*.ts` and `src/web/public/**` (`lint` only `src/**/*.ts`), and `.prettierignore` then exempts most of `src/web/public/*.js` (app.js, styles.css, **mobile.css**, index.html, upload.html, and 15 hand-formatted modules) plus `CLAUDE.md`. Those files are hand-formatted by design; `npm run check:public-assets` and `check:frontend-syntax` are what guard them (NUL bytes + JS syntax), not Prettier. Do not "fix" a file by adding it back to Prettier's scope. ## Common Gotchas @@ -171,14 +171,14 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph | **Attachments** | `src/attachment-registry.ts`, `attachment-magic`, `generated-artifact-attachments`, `session-attachment-history`, `document-preview-cache`, `document-thumbnailer`, `document-conversion-limiter`, `config/attachment-guard` | See Key Patterns | | **Plan** | `src/plan-orchestrator.ts`, `src/prompts/*.ts`, `src/templates/` (`claude-md.ts` + `case-template.md`) | `templates/` holds the CLAUDE.md scaffold generated into new cases | | **Web** | `src/web/server.ts` ★, `sse-events.ts`, `routes/*.ts` (25 modules + barrel; `session-routes.ts` ★), `route-helpers.ts`, `ports/*.ts`, `middleware/auth.ts`, `schemas.ts`, `self-update.ts`, `plan-usage-latest.ts`, `ws-connection-registry.ts`, `heic-jpeg-converter.ts` + `heic-jpeg-worker.ts` | | -| **Frontend** | `src/web/public/app.js` (~6.7K lines, core) + 32 modules + `sw.js` | See Frontend section for the load order, which is authoritative | +| **Frontend** | `src/web/public/app.js` (~6.9K lines, core) + 32 modules + `sw.js` (+ `voice-pcm-worklet.js`, fetched from JS, not in the load order) | See Frontend section for the load order, which is authoritative | | **Types** | `src/types/index.ts` (barrel) → 22 domain files; also `src/types.ts` root re-export | See `@fileoverview` in index.ts | ★ = Large, central file (>50KB) — read its `@fileoverview` first. All files have `@fileoverview` JSDoc — read that before diving in. Discovery aid: `grep -l '@fileoverview' src/web/routes/*.ts` lists all route modules; same grep works for `src/types/`, `src/web/public/*.js`. **Local packages**: `packages/xterm-zerolag-input/` (local echo overlay, single-source, see Gotchas). `packages/gesture-control/` (`codeman-gesture-control`, hand-tracking overlay source, built via `npm run build:gesture`). -**Config**: `src/config/` — 21 files, no barrel (`index.ts`) exists; import from the specific file. +**Config**: `src/config/` — 23 files plus the `cli-registry/` subdir, no barrel (`index.ts`) exists; import from the specific file. ⚠️ There are TWO `config/` directories: the repo-root `config/` holds tooling only (ESLint, knip, the vitest configs, `test-suites.ts`), while runtime config lives in `src/config/`. Throughout this file a bare `config/<name>.ts` in a code context means `src/config/<name>.ts`. **Utilities**: `src/utils/` — re-exported via index. Key: `CleanupManager`, `LRUMap` (⚠ NOT in the barrel — import from `./utils/lru-map.js` directly), `StaleExpirationMap`, `BufferAccumulator`, `stripAnsi`, `Debouncer`, `KeyedDebouncer`. Also: `claude-cli-resolver`/`opencode-cli-resolver`/`codex-cli-resolver`/`gemini-cli-resolver`/`antigravity-cli-resolver`/`pi-cli-resolver`/`grok-cli-resolver`/`deepseek-cli-resolver`/`omp-cli-resolver` (CLI path resolution, one per `SessionMode`, all nine sharing the lookup chain in `cli-executable-resolver`: server PATH, then that CLI's install dirs, then an interactive login shell LAST, since it is the only step that spawns anything and it is what finds nvm/Homebrew installs under a service manager's minimal PATH; ⚠ `pi-`, `grok-` and `deepseek-cli-resolver` additionally probe the binary's identity, since `pi` is a generic name, `grok` has npm squatters, and Debian ships an unrelated `dsh`), `file-query` (⚠ Files-panel search matcher, glob-by-two-pointer, never RegExp), `string-similarity` (fuzzy matching), `regex-patterns` (ANSI/token/spinner patterns), `assertNever` (exhaustive checks), `token-validation` (auth tokens), `nice-wrapper` (process priority), `shell-resolver` (⚠ resolves a real login shell for `mode: 'shell'`; the literal string `$SHELL` used to be expanded by the SERVER's shell, which is empty in a container), `event-loop-monitor` (a sync `execSync` freezes the port while the process stays alive, leaving no trace), `dependency-checker` + `dependency-report` (the `codeman doctor` probe engine, registry in `config/dependency-registry.ts`). @@ -381,7 +381,7 @@ Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. L ### API Routes -~228 handlers across 25 route files in `src/web/routes/`: system (56), sessions (34), cases (30), files (17), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), approvals (4), readmymind (4), me (2), teams (2), tab-layout (2), search (1), hooks (1), clipboard (1), status-telemetry (1), voice (1 + the `/ws/voice/stream` relay), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details. +~232 handlers across 25 route files in `src/web/routes/`: system (56), sessions (34), cases (34), files (17), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), approvals (4), readmymind (4), me (2), teams (2), tab-layout (2), search (1), hooks (1), clipboard (1), status-telemetry (1), voice (1 + the `/ws/voice/stream` relay), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details. **HTTP contract** (stable since 0.9.x, see `docs/versioning-policy.md`; full envelope/status/error-code/SSE spec in `docs/api-reference.md`): responses use the `ApiResponse<T>` envelope — `{ success: true, data? }` or `{ success: false, error, errorCode }` (`src/types/api.ts`). `/api/v1/*` is a versioned alias of `/api/*` (URL rewrite in `server.ts`). From dee674d3e2fc93553149c4c84f868deecb226e97 Mon Sep 17 00:00:00 2001 From: Codeman maintainer <noreply@anthropic.com> Date: Fri, 18 Sep 2026 13:41:45 +0200 Subject: [PATCH 20/28] fix(install): point the launcher-only caveat at the thing that resolves it The caveat #429 added ends with "see the docs above", and "the docs above" is CLI_DOCS[$i], which for DeepSeek is the upstream harness repo. Per docs/deepseek-integration.md the harness ships only the web, headless and base profiles, so following that link and running `npm install -g @deepseek-ai/dsh` leaves the reader exactly where the caveat is warning them about: a dsh that cannot drive a pane. What actually resolves it is Codeman's own Run dropdown, which offers "DeepSeek: add a terminal profile..." and installs one in a click. The new wording stays generic for any future launcherProfile entry, since Codeman is the thing being installed at all three call sites. Also flips one word in the generator: the comment said "see installCommandFor below" and that function is defined above it. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- install.sh | 2 +- scripts/generate-cli-catalog.mts | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/install.sh b/install.sh index 8b5672bc..29b6a13d 100755 --- a/install.sh +++ b/install.sh @@ -587,7 +587,7 @@ cli_catalog_print_install_hints() { elif [[ -n "${CLI_DOCS[$i]}" ]]; then echo -e " ${CLI_LABELS[$i]}: see ${CYAN}${CLI_DOCS[$i]}${NC}" if [[ "${CLI_LAUNCHER_ONLY[$i]}" == "1" ]]; then - echo -e " (its package installs a launcher only — it needs a profile that can drive a pane, see the docs above)" + echo -e " (installs a launcher only: it still needs a terminal profile, and Codeman's Run menu can add one)" fi fi done diff --git a/scripts/generate-cli-catalog.mts b/scripts/generate-cli-catalog.mts index e17df0ec..de0d7934 100644 --- a/scripts/generate-cli-catalog.mts +++ b/scripts/generate-cli-catalog.mts @@ -151,7 +151,7 @@ export function renderInstallShBlock(entries: CliEntry[] = STOCK_CLIS): string { labels.push(shQuote(entry.label)); enabled.push(entry.enabled ? '1' : '0'); // Parallel to CLI_IDS: 1 when this entry's install command installs a launcher rather - // than something that can drive a pane on its own (see installCommandFor below). Purely + // than something that can drive a pane on its own (see installCommandFor above). Purely // derived from discovery.launcherProfile — install.sh's hint printer reads this to add a // caveat instead of hardcoding which id it means. launcherOnly.push(entry.discovery.launcherProfile ? '1' : '0'); From ea5323d9908fd9ffe5651841076684627435c34a Mon Sep 17 00:00:00 2001 From: Codeman maintainer <noreply@anthropic.com> Date: Fri, 18 Sep 2026 13:42:44 +0200 Subject: [PATCH 21/28] test(input): pin the batched commit-plus-Enter ordering #441 fixes The unit harness proves WHICH candidate gets forwarded; the ordering is the half that shipped the bug, and only a real xterm shows it. The new browser case dispatches the character's keydown, its composed insertText and Enter's keydown in ONE page task, the shape an Android soft keyboard delivers through a single InputConnection transaction, and asserts what reaches the send path. Verified in both directions on this machine: with the drain in place the wire is `o\r`; with the drain removed (master's behaviour) it is `\r` and the character is gone entirely, because by the time the zero-delay timer runs xterm has emitted the `\r` and bumped the canonical counter past the candidate's snapshot, so the candidate stands down. The other four cases pass in both states. CLAUDE.md now names the decision point, what it costs (a keydown decides with less evidence than the timer did) and why that is safe for Enter, and says that the pin lives in a suite the CI gate does not run. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- CLAUDE.md | 2 +- ...rminal-keycode229-recovery.browser.test.ts | 83 ++++++++++++++++++- 2 files changed, 83 insertions(+), 2 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 7278266d..4690fcab 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -298,7 +298,7 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph ### Frontend -Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. Load order: `constants.js`(1) → `i18n.js`(1.5) → `mobile-handlers.js`(2) → `voice-input.js`(3) → `notification-manager.js`(4) → `keyboard-accessory.js`(5) → `input-cjk.js`(5.5) → `terminal-keycode229-recovery.js`(5.55) → `sanitize-html.js`(5.6) → `app.js`(6) → `tab-rail-resize.js`(6.5) → `terminal-ui.js`(7) → `respawn-ui.js`(8) → `ralph-panel.js`(9) → `orchestrator-panel.js`(9.5) → `cron-ui.js`(9.7) → `settings-ui.js`(10) → `panels-ui.js`(11) → `readmymind-ui.js`(11.3) → `ultracode-panel.js`(11.5) → `approvals-ui.js`(11.6) → `admin-ui.js`(11.7) → `session-ui.js`(12) → `webview-tabs.js`(12.5) → `mobile-overview.js`(12.55) → `home-sessions.js`(12.56) → `entrance-animations.js`(12.6) → `ralph-wizard.js`(13) → `api-client.js`(14) → `subagent-windows.js`(15) → `ultracode-windows.js`(15.5) → `session-lineage.js`(15.6) → `image-input.js`(16). `i18n.js` translates static + newly inserted application DOM while skipping terminal/response/file/user-name surfaces; `input-cjk.js` handles CJK IME composition via an always-visible textarea below the terminal (`window.cjkActive` blocks xterm's onData). `terminal-keycode229-recovery.js` forwards a committed `input` event that xterm's `_inputEvent` guard drops (Chrome-on-Android soft keyboards send `composed: true` after a keydown), and only when xterm emitted no canonical data for that keystroke. +Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. Load order: `constants.js`(1) → `i18n.js`(1.5) → `mobile-handlers.js`(2) → `voice-input.js`(3) → `notification-manager.js`(4) → `keyboard-accessory.js`(5) → `input-cjk.js`(5.5) → `terminal-keycode229-recovery.js`(5.55) → `sanitize-html.js`(5.6) → `app.js`(6) → `tab-rail-resize.js`(6.5) → `terminal-ui.js`(7) → `respawn-ui.js`(8) → `ralph-panel.js`(9) → `orchestrator-panel.js`(9.5) → `cron-ui.js`(9.7) → `settings-ui.js`(10) → `panels-ui.js`(11) → `readmymind-ui.js`(11.3) → `ultracode-panel.js`(11.5) → `approvals-ui.js`(11.6) → `admin-ui.js`(11.7) → `session-ui.js`(12) → `webview-tabs.js`(12.5) → `mobile-overview.js`(12.55) → `home-sessions.js`(12.56) → `entrance-animations.js`(12.6) → `ralph-wizard.js`(13) → `api-client.js`(14) → `subagent-windows.js`(15) → `ultracode-windows.js`(15.5) → `session-lineage.js`(15.6) → `image-input.js`(16). `i18n.js` translates static + newly inserted application DOM while skipping terminal/response/file/user-name surfaces; `input-cjk.js` handles CJK IME composition via an always-visible textarea below the terminal (`window.cjkActive` blocks xterm's onData). `terminal-keycode229-recovery.js` forwards a committed `input` event that xterm's `_inputEvent` guard drops (Chrome-on-Android soft keyboards send `composed: true` after a keydown), and only when xterm emitted no canonical data for that keystroke. ⚠️ **That decision is settled at the NEXT keydown as well as on its own zero-delay timer** (#441): the drain runs from xterm's custom key handler, which fires BEFORE xterm processes that key, so a soft keyboard that commits the last character and sends Enter in one InputConnection transaction puts the character on the wire ahead of the `\r`. On the timer alone that character is not merely late, it is LOST: xterm emits the `\r` first and bumps the canonical counter past the candidate's snapshot, so the candidate stands down (measured, `hell\r` where the user typed `hello`). The trade is that a keydown decides with less evidence than the timer did, since xterm's own keyCode-229 rescue has not run yet; that is safe for Enter, which clears the textarea so the pending diff emits nothing. Ordering is pinned by `test/terminal-keycode229-recovery.browser.test.ts`, which the CI gate does NOT run. **Entrance animations** (`entrance-animations.js`, all OFF by default): opt-in animations for the four things that appear when work starts, chosen per surface via `data-tab-anim` / `data-term-anim` / `data-win-anim` / `data-line-anim` on `<html>`. Defaults are the `legacy` theme, so an untouched install behaves exactly as before and every hook short-circuits on its first line. ⚠️ Tabs and connection lines are **destroyed mid-animation** on every re-render (`_fullRenderSessionTabs()` replaces the strip's innerHTML; `_updateConnectionLinesImmediate()` does `svg.innerHTML = ''`), so both are tracked by id and re-applied to the fresh element with a **negative `animation-delay`** to resume rather than restart. ⚠️ The terminal-pane styles may animate **transform / opacity / clip-path only**, xterm's FitAddon derives rows+cols from `getComputedStyle(parent).width/height`, so animating width/height/padding there would resize the PTY; `test/entrance-animations.test.ts` pins that property allowlist, plus the rule→keyframes→theme-option chain a style silently does nothing without. ⚠️ **`blur` is the ONE style that puts a `filter` on the terminal container**, against the standing rule, because every alternative was measured against a live xterm and does not work: a `backdrop-filter` veil on `::before` blurs perfectly while STATIC and Chrome silently drops the backdrop the moment ANY animation runs on that pseudo-element (the veil computes `blur(15.3px)` and the text behind it stays razor sharp), and driving the radius from rAF buys the same full-screen blur per frame plus main-thread work. The cost the rule exists to avoid is inherent to blurring a terminal, so the style buys it knowingly: opt-in, OFF by default, one ~520ms run per session open, class straight back off, `will-change` still unset. Worst-case price, headless SwiftShader with no GPU: frame deltas 16.7ms → 33.3ms for the run, against 16.7ms flat for `fade`. Do not generalise it — a second filtered terminal style needs its own measurement. ⚠️ The `blur` connection line animates `filter` too, so both kinds of line hold their glow in **`--line-glow`** and both of its keyframes say `blur(N) var(--line-glow)`: the function lists then match and interpolate, instead of the glow vanishing for the run and popping back (a lineage line's glow is a different colour entirely, set per element). Its 100% frame deliberately omits `opacity` so the endpoint comes from the element's own resting value — 0.9 subagent, 0.72 lineage, 0.95 working — which is what `line-enter-fade`'s hardcoded 0.9 gets wrong. ⚠️ Window styles other than `beam` transform the window, which moves the rect its connection line is aimed at; `beam` deliberately animates opacity/filter only so its line can draw toward a stable target. Persisted to its own `codeman:*Anim` localStorage keys (per-device, deliberately NOT in the `.strict()` `SettingsUpdateSchema`); picker in App Settings → Appearance, full per-surface lab at `?animlab=1`. diff --git a/test/terminal-keycode229-recovery.browser.test.ts b/test/terminal-keycode229-recovery.browser.test.ts index 8f980fc0..18b08d1e 100644 --- a/test/terminal-keycode229-recovery.browser.test.ts +++ b/test/terminal-keycode229-recovery.browser.test.ts @@ -9,7 +9,10 @@ * (stopPropagation, not stopImmediatePropagation) does not silence it; * - a `composed: true` insertText preceded by a keydown — the shape Chrome on * Android delivers — is dropped by xterm and recovered by us, exactly once; - * - a keystroke xterm DOES handle is delivered exactly once, not twice. + * - a keystroke xterm DOES handle is delivered exactly once, not twice; + * - a character committed in the SAME page task as Enter reaches the send + * path ahead of the `\r`, which is the ordering the zero-delay timer + * alone cannot produce. * * Browser-driven, so it is excluded from `npm run test:ci` like the other * Playwright suites. Run locally: @@ -146,6 +149,84 @@ describe('orphaned terminal input recovery wiring', () => { expect(second.sent.join('')).toBe('z'); }); + /** + * The batched shape an Android soft keyboard actually delivers when the user + * taps the last character and then Enter: the character's keydown, its + * `composed: true` insertText, and Enter's keydown all land in ONE page task, + * before any zero-delay timer can run. + * + * This is the ordering half of the fix, and the half the unit harness cannot + * reach: the unit tests prove WHICH candidate is forwarded, this proves WHEN. + * Resolving the pending candidate only on its 0 ms timer loses the character + * outright here, because by the time that timer runs xterm has already + * emitted the `\r` and bumped the canonical counter past the candidate's + * snapshot, so it stands down. Draining at the next keydown, from xterm's + * custom key handler (which runs before xterm processes that key), puts the + * character on the wire ahead of the `\r`. + */ + async function batchedCommitThenEnter(data: string) { + return page.evaluate(async (text) => { + const app = (window as any).app; + const textarea = document.querySelector('.xterm-helper-textarea') as HTMLTextAreaElement; + const originalSessionId = app.activeSessionId; + const originalLocalEcho = app._localEchoEnabled; + const originalSendInput = app._sendInputAsync; + const originalPendingInput = app._pendingInput; + const originalLastKeystrokeTime = app._lastKeystrokeTime; + const sent: string[] = []; + + try { + app.activeSessionId = 'cod388-browser-batched'; + app._localEchoEnabled = false; + app._pendingInput = ''; + app._lastKeystrokeTime = 0; + app._sendInputAsync = (_sessionId: string, chunk: string) => sent.push(chunk); + textarea.focus(); + + // One task, no awaits between the three dispatches. + const charDown = new KeyboardEvent('keydown', { + key: 'Unidentified', + bubbles: true, + cancelable: true, + composed: true, + }); + Object.defineProperties(charDown, { keyCode: { value: 65 }, which: { value: 65 } }); + textarea.dispatchEvent(charDown); + + textarea.value = text; + textarea.dispatchEvent( + new InputEvent('input', { data: text, inputType: 'insertText', bubbles: true, composed: true }) + ); + + const enterDown = new KeyboardEvent('keydown', { + key: 'Enter', + code: 'Enter', + bubbles: true, + cancelable: true, + composed: true, + }); + Object.defineProperties(enterDown, { keyCode: { value: 13 }, which: { value: 13 } }); + textarea.dispatchEvent(enterDown); + + await new Promise((resolve) => setTimeout(resolve, 80)); + return { wire: sent.join('') }; + } finally { + app.activeSessionId = originalSessionId; + app._localEchoEnabled = originalLocalEcho; + app._sendInputAsync = originalSendInput; + app._pendingInput = originalPendingInput; + app._lastKeystrokeTime = originalLastKeystrokeTime; + textarea.value = ''; + } + }, data); + } + + it('delivers a character committed in the same task as Enter BEFORE the carriage return', async () => { + const { wire } = await batchedCommitThenEnter('o'); + // Not '\r' (character lost, the defect) and not '\ro' (recovered too late). + expect(wire).toBe('o\r'); + }); + it('sends nothing for a keydown that produces no input event', async () => { const { sent } = await keystroke({ data: 'q', dispatchInput: false, keyCode: 65 }); expect(sent).toEqual([]); From bb8ada7e5fc5a3991db4392f972e99b4c8d53c12 Mon Sep 17 00:00:00 2001 From: Codeman maintainer <noreply@anthropic.com> Date: Fri, 18 Sep 2026 13:46:04 +0200 Subject: [PATCH 22/28] fix(reboot-restore): the merge-time items from the #442 review Seven things, none of which changes what the feature does. 1. The rebuilt Session dropped `nameSource`, so the constructor re-inferred it from the name: a session the user renamed by hand to something shaped like `w<n>-<case>` came back as `placeholder`, and with auto-naming on the next prompt overwrote their name. The route persists right after, so the loss went to disk. `restoreMuxSessions()` already passes it. 2. The already-live sets were snapshotted once before a loop that awaits a real `startInteractive()` per entry, so by the tenth entry the snapshot was tens of seconds old and a conversation resumed by hand from the Resume list in that window was invisible to it: two panes on one transcript, the exact thing the check exists to prevent. Both sets are now read per iteration, and the late case is spent rather than re-offered for the same reason the batch case is. 3. Auto-resume no longer re-arms the pre-reboot `autoResumeAt` on this path. The stamp predates the reboot and the pane is new, so honouring it meant one click had every restored session type `continue` into itself about a minute later, unattended, against the route header's own promise that a restored session comes back idle and disarmed. The setting stays ENABLED, so it re-arms on the next real limit message. A Codeman restart still re-arms from the stamp, because the limit footer will not reprint on its own; the new option exists only to tell the two paths apart. 4. `discardPartiallyBuiltSession()` now also calls `recordSessionStopped()` and `ralphTracker.fullReset()`, the two teardown steps `_doCleanupSession` performs that it was missing. Cosmetic, but a run left open reads as still going in the away digest. 5. A restored claude session gets `seedAgentSessionPreamble()` like both create paths, so the agent skill's bootstrap stays a two-line loader. 6. The heuristic's container comment was wrong in one direction and quiet about the real gap: after a genuine host reboot a containerized Codeman sees the host's short uptime and the banner does appear. What it cannot see is a container-only restart, which is where this would help most. 7. The banner is hidden in a solo window, which shows one session and has no tab strip to put restored ones in. Also reverts 17 of the 18 hunks in docs/api-reference.md, which were Prettier reformatting of prose the PR does not otherwise touch (docs/ is outside the format glob), keeping only the Reboot restore section and repairing the two continuation lines that reformat de-indented; renumbers reboot-restore-ui.js to @loadorder 11.65, since 11.7 is admin-ui.js, which loads after it; and gives the feature its CLAUDE.md entry plus a route test for the multi-user workspace-forbidden branch, the only new rule that had nothing behind it. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- CLAUDE.md | 10 +- docs/api-reference.md | 168 +++++++++++----------- src/reboot-restore.ts | 16 ++- src/web/ports/session-port.ts | 14 +- src/web/public/reboot-restore-ui.js | 2 +- src/web/public/styles.css | 4 + src/web/routes/reboot-restore-routes.ts | 54 ++++++- src/web/server.ts | 21 ++- test/routes/reboot-restore-routes.test.ts | 48 ++++++- 9 files changed, 228 insertions(+), 109 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 4690fcab..b3d765ac 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -170,8 +170,8 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph | **Search** | `src/search-service.ts` | Pure in-memory core for `GET /api/search` | | **Attachments** | `src/attachment-registry.ts`, `attachment-magic`, `generated-artifact-attachments`, `session-attachment-history`, `document-preview-cache`, `document-thumbnailer`, `document-conversion-limiter`, `config/attachment-guard` | See Key Patterns | | **Plan** | `src/plan-orchestrator.ts`, `src/prompts/*.ts`, `src/templates/` (`claude-md.ts` + `case-template.md`) | `templates/` holds the CLAUDE.md scaffold generated into new cases | -| **Web** | `src/web/server.ts` ★, `sse-events.ts`, `routes/*.ts` (25 modules + barrel; `session-routes.ts` ★), `route-helpers.ts`, `ports/*.ts`, `middleware/auth.ts`, `schemas.ts`, `self-update.ts`, `plan-usage-latest.ts`, `ws-connection-registry.ts`, `heic-jpeg-converter.ts` + `heic-jpeg-worker.ts` | | -| **Frontend** | `src/web/public/app.js` (~6.9K lines, core) + 32 modules + `sw.js` (+ `voice-pcm-worklet.js`, fetched from JS, not in the load order) | See Frontend section for the load order, which is authoritative | +| **Web** | `src/web/server.ts` ★, `sse-events.ts`, `routes/*.ts` (27 modules + barrel; `session-routes.ts` ★), `route-helpers.ts`, `ports/*.ts`, `middleware/auth.ts`, `schemas.ts`, `self-update.ts`, `plan-usage-latest.ts`, `ws-connection-registry.ts`, `heic-jpeg-converter.ts` + `heic-jpeg-worker.ts` | | +| **Frontend** | `src/web/public/app.js` (~6.9K lines, core) + 33 modules + `sw.js` (+ `voice-pcm-worklet.js`, fetched from JS, not in the load order) | See Frontend section for the load order, which is authoritative | | **Types** | `src/types/index.ts` (barrel) → 22 domain files; also `src/types.ts` root re-export | See `@fileoverview` in index.ts | ★ = Large, central file (>50KB) — read its `@fileoverview` first. All files have `@fileoverview` JSDoc — read that before diving in. Discovery aid: `grep -l '@fileoverview' src/web/routes/*.ts` lists all route modules; same grep works for `src/types/`, `src/web/public/*.js`. @@ -243,6 +243,8 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph **Hook events**: Claude Code hooks trigger via `/api/hook-event`. Key events: `permission_prompt`, `elicitation_dialog`, `elicitation_complete`, `elicitation_response`, `idle_prompt`, `stop`, `teammate_idle`, `task_completed`, `prompt_submitted` (UserPromptSubmit, #367: a Claude pane reports its live conversation id first-hand). See `src/hooks-config.ts`; upstream hook semantics mirrored in `docs/claude-code-hooks-reference.md`. ⚠️ **Every claude session INSTALLS the hooks block into its workspace** (`applyWorkspaceHooks` in hooks-config.ts → `ensureCodemanHooks`, an add-only merge that keeps a user's own handlers), from EVERY claude create path — both interactive routes, cron fires, legacy scheduled runs, the plan-orchestrator one-shots — and from `restoreMuxSessions()` for sessions recovered on server start (that boot sweep skips a workspace that no longer exists, so a deleted repo with a surviving tmux session is never resurrected as an empty dir). Before 2026-08-15 hooks were written ONLY when Codeman created the case DIRECTORY, so a linked case / cloned repo — where most sessions actually run — had no hooks at all and every hook-driven surface was silently dead there: an AskUserQuestion dialog blocked the pane while the tab and the phone overview both read a calm `idle`, with no Approvals Inbox item, no push, no definitive `stop`/`idle_prompt` for respawn and no `stop`/`blocked` for the wait endpoints. The escape hatch is the synced `workspaceHooksEnabled` setting (App Settings → Agents & CLIs → Claude, **default ON**); OFF restores the old behavior, where a Codeman block that is already there is still refreshed when stale (COD-91) but one is never added. ⚠️ Route the decision through `applyWorkspaceHooks` rather than calling `ensureCodemanHooks` at a new site, or the setting silently stops applying to that path. ⚠️ Claude Code RE-READS `settings.local.json`, so an already-running session starts firing hooks without a restart (measured 2026-08-15) — and the notification for a blocking dialog is delayed by Claude Code (~30s), so the alert trails the dialog. ⚠️ An AskUserQuestion / plan-selection dialog arrives as **`permission_prompt`**, not `elicitation_dialog` (that one is MCP elicitation), so it renders as the RED "needs you" alert, not the yellow idle one. +**Reboot restore** (#411/#442, `src/reboot-restore.ts` pure + `web/reboot-restore-registry.ts` + `routes/reboot-restore-routes.ts` + `reboot-restore-ui.js`): a host reboot takes the tmux server with it, so every pane dies and the board comes up empty with no explanation. At boot Codeman works out which sessions that reboot destroyed, holds the plan IN MEMORY (no new state file, and a server restart simply drops the offer), and the banner asks. ⚠️ **The heuristic decides whether to ASK, never whether to act**: two signals have to agree (the socket holds no panes at all while state still lists sessions, AND the host booted after the newest persisted activity), and a wrong yes costs one dismissable line rather than N CLI processes nobody asked for. ⚠️ Rebuilding is TAKE-then-build: entries leave the plan synchronously before the first `await` and the route is single-flighted per owner, so a double-click or a second device cannot put two panes on one conversation. Anything that never became a pane goes BACK on offer, with one deliberate exception, `already-live`, which unlike a missing workspace or a withdrawn grant cannot stop being true. ⚠️ Three things are re-checked at click time rather than trusted from boot (the owner's privilege grant, the workspace still being on disk, and the conversation not already being live), and the already-live sets are read FRESH per iteration rather than snapshotted: the loop awaits a real `startInteractive()` per entry, so a snapshot taken before it is tens of seconds stale by the tenth entry and would miss a conversation the user resumed by hand in that window. The confinement re-check is keyed on the entry's OWNER, never the caller, or an admin spending another user's entry is waved through by `isWorkingDirAllowed`. ⚠️ A rebuilt session comes back **attached, idle and disarmed**: the pane is NEW, so terminal scrollback is gone (the banner says so) while the conversation continues, respawn controllers and Ralph loops are never re-armed, and `reapplyPersistedSessionState(..., { rearmAutoResumeSchedule: false })` keeps auto-resume ENABLED but drops the pre-reboot `autoResumeAt` stamp, or one click has every restored session type `continue` into itself a minute later, unattended. That option exists only for this path; a Codeman restart still re-arms, because the limit footer will not reprint on its own. ⚠️ The rebuild passes `nameSource` through, or the constructor re-infers it from the name and a hand-renamed session shaped like `w<n>-<case>` comes back as `placeholder` for auto-naming to overwrite. ⚠️ Claude-mode only (others carry their conversation id in their own config object), and remote/docker sessions are never offered (`remote-or-docker`), because both need another host or container to be up. ⚠️ A failed rebuild is undone with `discardPartiallyBuiltSession()`, deliberately NOT `cleanupSession()`: the delete path would count the session's tokens into the lifetime totals, demote a pinned record to `stopped` (which this pass reads as an intentional kill, making the session permanently unrestorable) and recursively remove the WORKSPACE's `.claude-images`. ⚠️ `os.uptime()` reports the HOST's uptime, which a container shares, and that cuts both ways: after a genuine host reboot a containerized Codeman does see a short uptime and the banner works, but a container-only restart is invisible to it, which is the case where this would help most. Tests: `test/reboot-restore.test.ts`, `test/routes/reboot-restore-routes.test.ts`, `test/routes/reboot-restore-rebuild-failure.test.ts`, `test/discard-partially-built-session.test.ts`. + **Approvals Inbox** (cross-session queue of prompts waiting on a human; `approvalsInboxEnabled`, SYNCED, default OFF: every surface is opt-in; only the store and answer endpoints run regardless, so flipping it ON shows anything already pending): `web/approval-inbox.ts` is a `sessionWaits`-style singleton fed by `/api/hook-event`, holding at most ONE item per session (a new prompt supersedes), claude-mode only, in-memory. Cards are answered via `POST /api/approvals/:id/answer`, which sends a digit / Esc / idle-prompt text through `writeViaMux` (menu answers never carry `\r`). ⚠️ `option` digits are accepted ONLY when they match options parsed from the captured pane frame, and the answer path RE-CAPTURES the pane first (a dialog that no longer parses on screen means the keystroke would land in the composer, so refuse with 409). ⚠️ Resolution on the heuristic `working` signal ALONE is restricted to `idle` items; a permission/question item gets the pane-VERIFIED variant on that same signal (`resolveIfDialogGone()` → `verifyStillAnswerable()`), so the heuristic only decides when to LOOK and the screen decides the outcome. That is what clears a dialog answered in the terminal mid-turn; the other definitive signals are `stop`, `elicitation_complete`/`elicitation_response`, exit/delete, answer, supersede and the 12h TTL. ⚠️ **Viewing a session ACKNOWLEDGES its idle item, it does not resolve it** (`POST /api/approvals/session/:sessionId/viewed` → `acknowledgedAt` → `approval:updated`): the item stays pending (still answerable, still Read My Mind context) and only stops arming the yellow tab alert. That flag is what makes the clear durable, since the view-clears-idle rule used to live in one browser's memory and `seedApprovals()` re-armed the alert on the next reload while other devices never heard about it at all; the local half is `markIdleAlertSeen()` (app.js), called from BOTH `selectSession` paths, including the already-active early return, where a click could otherwise never clear the alert. ⚠️ **Only a HUMAN opening a session acknowledges**: `selectSession(id, { auto: true })` marks the three selections the APP makes (boot restore, a solo window opening its target, the fallback after the active session is closed) and skips the acknowledgement, so a page load cannot silently spend an alert the user never saw. The flag defaults to user-initiated, so an untagged call site fails toward acknowledging rather than toward an alert nothing can clear; `test/session-select-ack-gate.test.ts` pins both the gate and the tagged call sites. Idle-only by construction (`acknowledge()` defaults to `['idle']`): looking at a permission/question dialog does not answer it. ⚠️ Same rule on the input path: `_ackDelivery` (app.js) spends the IDLE alert only, via that same `markIdleAlertSeen()`. It used to `clearPendingHooks(sessionId)` with no kind, so one keystroke wiped a RED alert on that device while the dialog was still up, the other devices stayed red, and a reload re-seeded it. ⚠️ Claude Code fires no "permission answered" hook (only `elicitation_complete`/`elicitation_response`, i.e. the question flavor), so an answered-in-the-terminal dialog would otherwise sit pending until `stop`: `GET /api/approvals` therefore runs a **staleness sweep** over the caller's own items via `verifyStillAnswerable()`, which is deliberately the conservative check the answer path uses (only an item whose ORIGINAL frame parsed options can be dropped, so an unreadable capture keeps the alert rather than losing a live one). ⚠️ **`applyCapture()` is therefore ADD-ONLY for `options`**: a re-capture that parses nothing must never erase a parse an earlier one found. Claude Code delays the Notification hook behind the dialog (measured 6s, documented ~30s), so the 600ms re-capture routinely lands on a frame the user has ALREADY answered; clearing the field there made the item permanently unsweepable, because `verifyStillAnswerable()` reads a MISSING `options` as "we never could read this dialog" and keeps such items answerable by design. The red "needs you" then survived every sweep AND every page reload, went away only on `stop` (2026-08-20: a confirmed question left a tab flowing red for ~8 minutes while the turn ran on), and the stale card still accepted an answer, typing a bare `1` into a composer with no dialog under it. Pinned by `test/approval-inbox.test.ts`. ⚠️ A frame that parses no options is CONCLUSIVE in exactly two cases, and the second one closes the late-hook hole: the item once parsed options (they cannot vanish while the dialog is up), or the frame shows Claude actively running a turn. A modal dialog BLOCKS the turn, so the two cannot coexist — measured on v2.1.237, a live-dialog frame carries neither the `… (13s` timer NOR the `esc to interrupt` footer, which the dialog replaces with `Enter to select · ↑/↓ to navigate · Esc to cancel`. Anything else stays answerable, so an unreadable capture still keeps the alert. That second signal is reached by a delayed staleness pass (`STALE_CHECK_DELAY_MS`, 3s) scheduled alongside the re-capture, because a prompt answered BEFORE the hook lands creates an item whose FIRST capture already has no dialog in it: nothing ever parsed, `stop` may have fired already, and the alert then outlived reloads until the 12h TTL. ⚠️ That pass must stay comfortably LATER than `RECAPTURE_DELAY_MS`, whose whole reason for existing is that the hook can beat Ink to the screen — resolving inside the paint window would clear the alert for a dialog that was about to appear. The frontend seeds from `GET /api/approvals` in `handleInit` **regardless of the setting**: the seed re-arms the tab-alert state machine (`setPendingHook`) unconditionally, and only populating `this.approvals` (the inbox surfaces) is gated — seeding used to be gated wholesale, which left a reloaded page with NO red tab while a permission dialog sat blocking a session (2026-08-15); `_onApprovalResolved` clears the pending-hook alert unconditionally for the same reason. ⚠️ The red/yellow tab alert itself is a STEADY border/background/dot with a pulse on top: the original keyframes swung to transparent at 0%/100%, so half of every cycle looked like a normal tab. Push Approve/Deny buttons stay gated on the setting (`sendPushNotifications` strips `actions`/`approvalId` when OFF) and are answered from `sw.js` directly so they work with no tab open. Surfaces (all gated on the setting): header bell (marker-hidden until count > 0, phones never show it) + drawer (`approvals-ui.js`), phone overview NEEDS YOU answer strips (`mobile-overview.js`). Design: `docs/approvals-inbox-plan.md`. **Read My Mind intent profiles** (phase 1 of `docs/readmymind-plan.md`; `readMyMindEnabled`, SYNCED, default OFF): per-CASE profiles (user-stated `goals` + the user's recent real prompts), keyed by owner + realpath(workingDir) so they survive `/clear`/respawns and multi-user scoping is structural. Capture rides the transcript (`transcript:user_prompt` from `transcript-watcher.ts`), NOT the input paths: `POST /input` sees only programmatic prompts and the WS channel is raw keystrokes. The listener lives inside `startTranscriptWatcher()`'s `if (!watcher)` block (outside it would duplicate per hook event) and is claude-only + gated on the setting per event. Store: `src/intent-store.ts` singleton, `intents.json` written 0600 tmp+rename (prompts can contain secrets; never fed to `/api/search`). Endpoints: GET/PUT/DELETE `/api/sessions/:id/intent` + POST `/api/sessions/:id/readmymind` (`readmymind-routes.ts`, ownership via `findSessionOrFail` WITH `req`; registrations stay the bare `app.<method>('path')` shape, the endpoints.md drift scanner cannot see generics). **Phase 2 (predictor + 🧠 button)**: `readmymind-context.ts` is the PURE budgeted assembler (9 ranked sources, drop order siblings→away→workspace→tools, sections 1-4 truncate only); IO lives in `readmymind-collectors.ts` (transcript TAIL read — the live watcher keeps only a 500-char snippet — + git signals, skipped for remote-SSH cases) and the route; `readmymind-predictor.ts` reuses the AiCheckerBase spawn mechanics standalone (verdict-shaped base vs freeform JSON) as a mutable singleton routes call and tests stub. Claude-mode only (400), one in flight per session (409 CONFLICT), model = `readMyMindModel` setting defaulting to `AI_CHECK_MODEL` (opus, decided). Frontend `readmymind-ui.js`: header 🧠 marker-hidden (`btn-readmymind--hidden`) until the setting is ON; phones hide it in mobile.css and get a keyboard-accessory 🧠 key instead (ships in BOTH bar templates, revealed by the `rmm-enabled` class on the BAR element — setMode() rebuilds button innerHTML, so per-key state would be wiped; synced at init + every `applyHeaderVisibilitySettings()`). Alternate suggestions render as tappable rows that swap into the editable field without losing edits; Rethink rejects the whole shown set and carries the optional steer note (`#readMyMindSteer`, sent as `steer`, shown in ready + empty-result phases, cleared on each open). Suggestions render via value/`textContent` ONLY and Send/Insert go through `POST /input` (server-side, so the sendEnterKey/local-echo trap does not apply) — nothing auto-sends, ever. User guide: `docs/readmymind.md`. @@ -298,7 +300,7 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph ### Frontend -Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. Load order: `constants.js`(1) → `i18n.js`(1.5) → `mobile-handlers.js`(2) → `voice-input.js`(3) → `notification-manager.js`(4) → `keyboard-accessory.js`(5) → `input-cjk.js`(5.5) → `terminal-keycode229-recovery.js`(5.55) → `sanitize-html.js`(5.6) → `app.js`(6) → `tab-rail-resize.js`(6.5) → `terminal-ui.js`(7) → `respawn-ui.js`(8) → `ralph-panel.js`(9) → `orchestrator-panel.js`(9.5) → `cron-ui.js`(9.7) → `settings-ui.js`(10) → `panels-ui.js`(11) → `readmymind-ui.js`(11.3) → `ultracode-panel.js`(11.5) → `approvals-ui.js`(11.6) → `admin-ui.js`(11.7) → `session-ui.js`(12) → `webview-tabs.js`(12.5) → `mobile-overview.js`(12.55) → `home-sessions.js`(12.56) → `entrance-animations.js`(12.6) → `ralph-wizard.js`(13) → `api-client.js`(14) → `subagent-windows.js`(15) → `ultracode-windows.js`(15.5) → `session-lineage.js`(15.6) → `image-input.js`(16). `i18n.js` translates static + newly inserted application DOM while skipping terminal/response/file/user-name surfaces; `input-cjk.js` handles CJK IME composition via an always-visible textarea below the terminal (`window.cjkActive` blocks xterm's onData). `terminal-keycode229-recovery.js` forwards a committed `input` event that xterm's `_inputEvent` guard drops (Chrome-on-Android soft keyboards send `composed: true` after a keydown), and only when xterm emitted no canonical data for that keystroke. ⚠️ **That decision is settled at the NEXT keydown as well as on its own zero-delay timer** (#441): the drain runs from xterm's custom key handler, which fires BEFORE xterm processes that key, so a soft keyboard that commits the last character and sends Enter in one InputConnection transaction puts the character on the wire ahead of the `\r`. On the timer alone that character is not merely late, it is LOST: xterm emits the `\r` first and bumps the canonical counter past the candidate's snapshot, so the candidate stands down (measured, `hell\r` where the user typed `hello`). The trade is that a keydown decides with less evidence than the timer did, since xterm's own keyCode-229 rescue has not run yet; that is safe for Enter, which clears the textarea so the pending diff emits nothing. Ordering is pinned by `test/terminal-keycode229-recovery.browser.test.ts`, which the CI gate does NOT run. +Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. Load order: `constants.js`(1) → `i18n.js`(1.5) → `mobile-handlers.js`(2) → `voice-input.js`(3) → `notification-manager.js`(4) → `keyboard-accessory.js`(5) → `input-cjk.js`(5.5) → `terminal-keycode229-recovery.js`(5.55) → `sanitize-html.js`(5.6) → `app.js`(6) → `tab-rail-resize.js`(6.5) → `terminal-ui.js`(7) → `respawn-ui.js`(8) → `ralph-panel.js`(9) → `orchestrator-panel.js`(9.5) → `cron-ui.js`(9.7) → `settings-ui.js`(10) → `panels-ui.js`(11) → `readmymind-ui.js`(11.3) → `ultracode-panel.js`(11.5) → `approvals-ui.js`(11.6) → `reboot-restore-ui.js`(11.65) → `admin-ui.js`(11.7) → `session-ui.js`(12) → `webview-tabs.js`(12.5) → `mobile-overview.js`(12.55) → `home-sessions.js`(12.56) → `entrance-animations.js`(12.6) → `ralph-wizard.js`(13) → `api-client.js`(14) → `subagent-windows.js`(15) → `ultracode-windows.js`(15.5) → `session-lineage.js`(15.6) → `image-input.js`(16). `i18n.js` translates static + newly inserted application DOM while skipping terminal/response/file/user-name surfaces; `input-cjk.js` handles CJK IME composition via an always-visible textarea below the terminal (`window.cjkActive` blocks xterm's onData). `terminal-keycode229-recovery.js` forwards a committed `input` event that xterm's `_inputEvent` guard drops (Chrome-on-Android soft keyboards send `composed: true` after a keydown), and only when xterm emitted no canonical data for that keystroke. ⚠️ **That decision is settled at the NEXT keydown as well as on its own zero-delay timer** (#441): the drain runs from xterm's custom key handler, which fires BEFORE xterm processes that key, so a soft keyboard that commits the last character and sends Enter in one InputConnection transaction puts the character on the wire ahead of the `\r`. On the timer alone that character is not merely late, it is LOST: xterm emits the `\r` first and bumps the canonical counter past the candidate's snapshot, so the candidate stands down (measured, `hell\r` where the user typed `hello`). The trade is that a keydown decides with less evidence than the timer did, since xterm's own keyCode-229 rescue has not run yet; that is safe for Enter, which clears the textarea so the pending diff emits nothing. Ordering is pinned by `test/terminal-keycode229-recovery.browser.test.ts`, which the CI gate does NOT run. **Entrance animations** (`entrance-animations.js`, all OFF by default): opt-in animations for the four things that appear when work starts, chosen per surface via `data-tab-anim` / `data-term-anim` / `data-win-anim` / `data-line-anim` on `<html>`. Defaults are the `legacy` theme, so an untouched install behaves exactly as before and every hook short-circuits on its first line. ⚠️ Tabs and connection lines are **destroyed mid-animation** on every re-render (`_fullRenderSessionTabs()` replaces the strip's innerHTML; `_updateConnectionLinesImmediate()` does `svg.innerHTML = ''`), so both are tracked by id and re-applied to the fresh element with a **negative `animation-delay`** to resume rather than restart. ⚠️ The terminal-pane styles may animate **transform / opacity / clip-path only**, xterm's FitAddon derives rows+cols from `getComputedStyle(parent).width/height`, so animating width/height/padding there would resize the PTY; `test/entrance-animations.test.ts` pins that property allowlist, plus the rule→keyframes→theme-option chain a style silently does nothing without. ⚠️ **`blur` is the ONE style that puts a `filter` on the terminal container**, against the standing rule, because every alternative was measured against a live xterm and does not work: a `backdrop-filter` veil on `::before` blurs perfectly while STATIC and Chrome silently drops the backdrop the moment ANY animation runs on that pseudo-element (the veil computes `blur(15.3px)` and the text behind it stays razor sharp), and driving the radius from rAF buys the same full-screen blur per frame plus main-thread work. The cost the rule exists to avoid is inherent to blurring a terminal, so the style buys it knowingly: opt-in, OFF by default, one ~520ms run per session open, class straight back off, `will-change` still unset. Worst-case price, headless SwiftShader with no GPU: frame deltas 16.7ms → 33.3ms for the run, against 16.7ms flat for `fade`. Do not generalise it — a second filtered terminal style needs its own measurement. ⚠️ The `blur` connection line animates `filter` too, so both kinds of line hold their glow in **`--line-glow`** and both of its keyframes say `blur(N) var(--line-glow)`: the function lists then match and interpolate, instead of the glow vanishing for the run and popping back (a lineage line's glow is a different colour entirely, set per element). Its 100% frame deliberately omits `opacity` so the endpoint comes from the element's own resting value — 0.9 subagent, 0.72 lineage, 0.95 working — which is what `line-enter-fade`'s hardcoded 0.9 gets wrong. ⚠️ Window styles other than `beam` transform the window, which moves the rect its connection line is aimed at; `beam` deliberately animates opacity/filter only so its line can draw toward a stable target. Persisted to its own `codeman:*Anim` localStorage keys (per-device, deliberately NOT in the `.strict()` `SettingsUpdateSchema`); picker in App Settings → Appearance, full per-surface lab at `?animlab=1`. @@ -381,7 +383,7 @@ Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. L ### API Routes -~232 handlers across 25 route files in `src/web/routes/`: system (56), sessions (34), cases (34), files (17), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), approvals (4), readmymind (4), me (2), teams (2), tab-layout (2), search (1), hooks (1), clipboard (1), status-telemetry (1), voice (1 + the `/ws/voice/stream` relay), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details. +~233 handlers across 27 route files in `src/web/routes/`: system (56), sessions (34), cases (34), files (17), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), approvals (4), readmymind (4), custom-model (5), reboot-restore (3), me (2), teams (2), tab-layout (2), search (1), hooks (1), clipboard (1), status-telemetry (1), voice (1 + the `/ws/voice/stream` relay), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details. **HTTP contract** (stable since 0.9.x, see `docs/versioning-policy.md`; full envelope/status/error-code/SSE spec in `docs/api-reference.md`): responses use the `ApiResponse<T>` envelope — `{ success: true, data? }` or `{ success: false, error, errorCode }` (`src/types/api.ts`). `/api/v1/*` is a versioned alias of `/api/*` (URL rewrite in `server.ts`). diff --git a/docs/api-reference.md b/docs/api-reference.md index 4bca9768..53b38e8a 100644 --- a/docs/api-reference.md +++ b/docs/api-reference.md @@ -66,17 +66,17 @@ The single source of truth is `ErrorStatus` / `httpStatusForErrorCode()` in `src/types/api.ts`. Clients should branch on `errorCode` (stable) and may rely on the HTTP status. -| `errorCode` | HTTP | Meaning | -| ------------------ | ---- | --------------------------------------------------- | -| `INVALID_INPUT` | 400 | Malformed request / failed validation | -| `UNAUTHORIZED` | 401 | Authentication required or failed | -| `NOT_FOUND` | 404 | Resource does not exist | -| `SESSION_BUSY` | 409 | Session is busy | -| `CONFLICT` | 409 | Conflicts with current state (e.g. already running) | -| `ALREADY_EXISTS` | 409 | Resource already exists | -| `OPERATION_FAILED` | 422 | Well-formed but could not be completed | -| `RATE_LIMITED` | 429 | Too many requests | -| `INTERNAL_ERROR` | 500 | Unexpected server error | +| `errorCode` | HTTP | Meaning | +|-------------|------|---------| +| `INVALID_INPUT` | 400 | Malformed request / failed validation | +| `UNAUTHORIZED` | 401 | Authentication required or failed | +| `NOT_FOUND` | 404 | Resource does not exist | +| `SESSION_BUSY` | 409 | Session is busy | +| `CONFLICT` | 409 | Conflicts with current state (e.g. already running) | +| `ALREADY_EXISTS` | 409 | Resource already exists | +| `OPERATION_FAILED` | 422 | Well-formed but could not be completed | +| `RATE_LIMITED` | 429 | Too many requests | +| `INTERNAL_ERROR` | 500 | Unexpected server error | Adding a new error code is non-breaking; removing or renaming one is a major change. @@ -87,10 +87,10 @@ exist because SSE is Codeman's only other "tell me when" channel, and an agent driving the API from a shell tool cannot practically hold a stream and parse events inline. -| Call | Blocks until | -| --------------------------------------------- | -------------------------------------------------- | -| `GET /api/v1/sessions/:id/wait` | one of a set of lifecycle signals fires | -| `GET /api/v1/sessions/:id/wait-output` | a literal string appears in the session's output | +| Call | Blocks until | +|------|--------------| +| `GET /api/v1/sessions/:id/wait` | one of a set of lifecycle signals fires | +| `GET /api/v1/sessions/:id/wait-output` | a literal string appears in the session's output | | `POST /api/v1/sessions/:id/input` with `wait` | the input is delivered **and then** a signal fires | `POST .../input` with `wait` is not the same as a `POST` followed by a separate @@ -140,13 +140,13 @@ contract is a **marker unique to each call** (`MARK="DONE_$RANDOM"`, send ### Signals -| Signal | Source | Actually fires for | -| --------- | -------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `idle` | the session's own `idle` event | `claude`: yes, on ❯-prompt detection after activity. `shell`: **once only**, ~500 ms after start, and never again. External CLIs: not guaranteed (they render their own TUIs and readiness is output stabilization) | -| `working` | the session's own `working` event | `claude` only in practice (spinner and work-keyword detection are Claude output formats) | -| `stop` | the Claude Code `stop` hook, the definitive end-of-turn signal | `claude` only | -| `blocked` | a `permission_prompt` or `elicitation_dialog` hook | `claude` only, and rarer than it looks: see below | -| `exit` | no process is behind the session | every mode | +| Signal | Source | Actually fires for | +|--------|--------|--------------------| +| `idle` | the session's own `idle` event | `claude`: yes, on ❯-prompt detection after activity. `shell`: **once only**, ~500 ms after start, and never again. External CLIs: not guaranteed (they render their own TUIs and readiness is output stabilization) | +| `working` | the session's own `working` event | `claude` only in practice (spinner and work-keyword detection are Claude output formats) | +| `stop` | the Claude Code `stop` hook, the definitive end-of-turn signal | `claude` only | +| `blocked` | a `permission_prompt` or `elicitation_dialog` hook | `claude` only, and rarer than it looks: see below | +| `exit` | no process is behind the session | every mode | `stop` is the signal to orchestrate on where it exists; `idle` is a heuristic fallback that can flap mid-turn when a spinner pauses. The default set when `until` @@ -156,12 +156,12 @@ can no longer happen). On a `claude` worker, prefer an explicit `until=stop,exit once the session is up: the default set's `idle` also resolves on a spinner pause, and on a fresh session the **startup** `idle` (emitted when the CLI first comes up) can land inside your first wait window and report a turn that never ran. Measured: -a session parked on the trust dialog emits no _further_ `idle`, so it is the +a session parked on the trust dialog emits no *further* `idle`, so it is the startup transition, not the dialog, that produces the false success below. ⚠️ **`exit` means "nothing is running", which includes "not started yet".** The server answers from `pid === null` plus a mux-layer pane-death probe, and that -covers a session that exited — including a worker that died _inside_ its tmux pane +covers a session that exited — including a worker that died *inside* its tmux pane while the local attach client (and therefore `pid`) lives on — one that was detached, and one that was **created but never started**. So the first wait after `POST /api/v1/sessions` returns `{"signal":"exit","immediate":true}` in @@ -184,7 +184,7 @@ blocked, and polling `blocked` alone will sit at its timeout. ⚠️ **On a `shell` session, only `exit` and marker-matching are dependable.** A shell session emits its one `idle` at startup and then stays `status: "idle"` forever, -whatever the pane is doing, so it never emits a _transition_. Since send-and-wait +whatever the pane is doing, so it never emits a *transition*. Since send-and-wait requires a transition (and so does `fresh=1`), both can only time out there: a documented default `wait` on a shell worker running `sleep 4` times out at the full 25 s. Synchronize hook-less sessions with `wait-output` and a unique marker @@ -218,11 +218,11 @@ with `from=buffer` keeps matching long after the dialog is gone. A worked versio ### `GET /api/v1/sessions/:id/wait` -| Param | Type | Default | Notes | -| --------- | -------------------------------------------------------- | ---------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `until` | comma-separated list of `idle,working,stop,blocked,exit` | `stop,idle,exit` | resolves on the first to fire. An unknown token is a `400` naming it, never a silent fallback | -| `timeout` | positive integer ms | `60000` | **validated first, clamped second.** `0`, a negative value and a fractional value are all `400`s, not clamps; a valid value outside `[1000, 600000]` is clamped and echoed as `wait.timeoutMs` | -| `fresh` | `0` \| `1` \| `false` \| `true` | `0` | `1` requires an actual transition, ignoring the state at call time | +| Param | Type | Default | Notes | +|-------|------|---------|-------| +| `until` | comma-separated list of `idle,working,stop,blocked,exit` | `stop,idle,exit` | resolves on the first to fire. An unknown token is a `400` naming it, never a silent fallback | +| `timeout` | positive integer ms | `60000` | **validated first, clamped second.** `0`, a negative value and a fractional value are all `400`s, not clamps; a valid value outside `[1000, 600000]` is clamped and echoed as `wait.timeoutMs` | +| `fresh` | `0` \| `1` \| `false` \| `true` | `0` | `1` requires an actual transition, ignoring the state at call time | ```bash curl -s "$API/api/v1/sessions/$SID/wait?until=stop,exit&timeout=60000" @@ -239,12 +239,12 @@ a plain signal wait, so check the endpoint path before blaming the parameters. ### `GET /api/v1/sessions/:id/wait-output` -| Param | Type | Default | Notes | -| --------- | ------------------------------- | -------- | ----------------------------------------------------------------------------------------------------------- | -| `match` | literal string, 1 to 200 chars | required | substring match against the PTY stream with ANSI escapes stripped. A match spanning two PTY chunks is found | -| `nocase` | `0` \| `1` \| `false` \| `true` | `0` | case-insensitive compare. The returned snippet keeps the terminal's original casing | -| `from` | `now` \| `buffer` | `now` | `buffer` scans the tail of the existing terminal buffer (bounded, 256 KB by default) before blocking | -| `timeout` | positive integer ms | `60000` | same validation and clamp as `/wait` | +| Param | Type | Default | Notes | +|-------|------|---------|-------| +| `match` | literal string, 1 to 200 chars | required | substring match against the PTY stream with ANSI escapes stripped. A match spanning two PTY chunks is found | +| `nocase` | `0` \| `1` \| `false` \| `true` | `0` | case-insensitive compare. The returned snippet keeps the terminal's original casing | +| `from` | `now` \| `buffer` | `now` | `buffer` scans the tail of the existing terminal buffer (bounded, 256 KB by default) before blocking | +| `timeout` | positive integer ms | `60000` | same validation and clamp as `/wait` | **Matching is literal, never a pattern.** A `regex` parameter is rejected with a `400` rather than ignored, so a caller that assumed otherwise finds out immediately @@ -296,10 +296,10 @@ hand-written query string decodes to a space. Two optional fields on the existing endpoint: -| Field | Type | Notes | -| ------------- | ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| `wait` | `true` or the same comma grammar as `until` | `true` means the default signal set. Omitted keeps the historical fire-and-forget behavior, unchanged. `null`, `false` and an empty string are all read as **absent**, not as an error and not as "wait for the default" | -| `waitTimeout` | positive integer ms | same validation **and** clamp as `timeout`: `0`, a negative and a fractional value are `400`s, anything valid is clamped into `[1000, 600000]` and echoed as `wait.timeoutMs` | +| Field | Type | Notes | +|-------|------|-------| +| `wait` | `true` or the same comma grammar as `until` | `true` means the default signal set. Omitted keeps the historical fire-and-forget behavior, unchanged. `null`, `false` and an empty string are all read as **absent**, not as an error and not as "wait for the default" | +| `waitTimeout` | positive integer ms | same validation **and** clamp as `timeout`: `0`, a negative and a fractional value are `400`s, anything valid is clamped into `[1000, 600000]` and echoed as `wait.timeoutMs` | Both are `nullish`, so an explicit `null` from `JSON.stringify` is accepted as "absent" rather than failing validation. That is deliberate: `.optional()` would @@ -330,24 +330,16 @@ All three nest the wait result under `data.wait`, so one client helper works aga any of them: ```json -{ - "success": true, - "data": { - "sessionId": "28325fd3-caa7-4178-82bf-87dfebf0f464", - "status": "idle", - "limitPaused": false, - "wait": { - "signal": "stop", - "until": ["stop", "idle", "exit"], - "timedOut": false, - "immediate": false, - "ended": false, - "aborted": false, - "waitedMs": 8421, - "timeoutMs": 60000 - } +{ "success": true, "data": { + "sessionId": "28325fd3-caa7-4178-82bf-87dfebf0f464", + "status": "idle", + "limitPaused": false, + "wait": { + "signal": "stop", "until": ["stop", "idle", "exit"], + "timedOut": false, "immediate": false, "ended": false, "aborted": false, + "waitedMs": 8421, "timeoutMs": 60000 } -} +}} ``` `POST .../input` returns the same `wait` object alongside `delivered`, `duplicate`, @@ -361,21 +353,21 @@ redelivery (harmless, the turn it refers to may be long over), while with client that reads `delivered === false` as "duplicate" silently treats a failed send as a success. -| Field | Type | Meaning | -| ---------------- | ---------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `wait.signal` | signal \| `null` | the signal that fired (`/wait` and `/input` only) | -| `wait.until` | array of signals | what the server actually waited on, after narrowing the default set for the session's mode (`/wait` and `/input` only) | -| `wait.matched` | boolean | the string appeared (`/wait-output` only) | -| `wait.match` | string | the literal that was searched for (`/wait-output` only) | -| `wait.snippet` | string \| `null` | bounded window of output around the match, blank runs collapsed for readability (`/wait-output` only) | -| `wait.timedOut` | boolean | the wait hit its timeout. Still a `200` | -| `wait.immediate` | boolean | the condition already held at call time, so nothing was waited for (`waitedMs` is 0) | -| `wait.ended` | boolean | the session went away (deleted or torn down) before the condition was met | -| `wait.aborted` | boolean | the client hung up, so the waiter was released without resolving — and by that definition a client never reads `true`. When the **server** abandons a wait itself (send-and-wait against a session with no PTY), it answers in about a millisecond with `ended: true`, `delivered: false`, `duplicate: false` and `aborted: false`: `delivered`/`ended` carry that story, and `aborted` stays the transport flag. Present for completeness; treat a `true` as "this wait answered nothing", never as an outcome | -| `wait.waitedMs` | number | wall-clock ms actually spent waiting | -| `wait.timeoutMs` | number | the timeout **after clamping**, which is what was applied | -| `status` | `SessionStatus` | the session's status after the wait, so a caller that timed out still learns where things stand | -| `limitPaused` | boolean | the session is paused on a usage limit and will emit nothing until its reset, so a timeout here is expected rather than a stall worth retrying hard | +| Field | Type | Meaning | +|-------|------|---------| +| `wait.signal` | signal \| `null` | the signal that fired (`/wait` and `/input` only) | +| `wait.until` | array of signals | what the server actually waited on, after narrowing the default set for the session's mode (`/wait` and `/input` only) | +| `wait.matched` | boolean | the string appeared (`/wait-output` only) | +| `wait.match` | string | the literal that was searched for (`/wait-output` only) | +| `wait.snippet` | string \| `null` | bounded window of output around the match, blank runs collapsed for readability (`/wait-output` only) | +| `wait.timedOut` | boolean | the wait hit its timeout. Still a `200` | +| `wait.immediate` | boolean | the condition already held at call time, so nothing was waited for (`waitedMs` is 0) | +| `wait.ended` | boolean | the session went away (deleted or torn down) before the condition was met | +| `wait.aborted` | boolean | the client hung up, so the waiter was released without resolving — and by that definition a client never reads `true`. When the **server** abandons a wait itself (send-and-wait against a session with no PTY), it answers in about a millisecond with `ended: true`, `delivered: false`, `duplicate: false` and `aborted: false`: `delivered`/`ended` carry that story, and `aborted` stays the transport flag. Present for completeness; treat a `true` as "this wait answered nothing", never as an outcome | +| `wait.waitedMs` | number | wall-clock ms actually spent waiting | +| `wait.timeoutMs` | number | the timeout **after clamping**, which is what was applied | +| `status` | `SessionStatus` | the session's status after the wait, so a caller that timed out still learns where things stand | +| `limitPaused` | boolean | the session is paused on a usage limit and will emit nothing until its reset, so a timeout here is expected rather than a stall worth retrying hard | Read the outcome by discriminator, in this order: @@ -398,12 +390,12 @@ read the timeout as "the worker is wedged" and kill a session that was working f ### Errors -| `errorCode` | HTTP | When | -| --------------- | ---- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `INVALID_INPUT` | 400 | unknown `until` / `wait` token; `stop` or `blocked` requested explicitly on a mode that installs no hooks (the message names the mode); `regex=` on `/wait-output`; `match` outside 1 to 200 chars; a non-numeric `timeout` | -| `NOT_FOUND` | 404 | no such session, or one this caller does not own | -| `SESSION_BUSY` | 409 | this session's waiter cap is full | -| `RATE_LIMITED` | 429 | a per-owner or process-wide waiter cap is full. Retry later; the session you named is not the problem | +| `errorCode` | HTTP | When | +|-------------|------|------| +| `INVALID_INPUT` | 400 | unknown `until` / `wait` token; `stop` or `blocked` requested explicitly on a mode that installs no hooks (the message names the mode); `regex=` on `/wait-output`; `match` outside 1 to 200 chars; a non-numeric `timeout` | +| `NOT_FOUND` | 404 | no such session, or one this caller does not own | +| `SESSION_BUSY` | 409 | this session's waiter cap is full | +| `RATE_LIMITED` | 429 | a per-owner or process-wide waiter cap is full. Retry later; the session you named is not the problem | The two capacity codes are deliberately different. A process-wide cap reported as `SESSION_BUSY` would tell the caller to switch sessions, which cannot help. The @@ -454,9 +446,9 @@ Design: [`approvals-inbox-plan.md`](approvals-inbox-plan.md). - `GET /api/v1/approvals` → `{ approvals: ApprovalItem[] }`, oldest first, ownership-scoped in multi-user mode. `ApprovalItem`: `{ id, sessionId, -sessionName, kind: 'permission'|'question'|'idle', createdAt, toolName?, -toolSummary?, message?, cwd?, context?, options?: {n, label}[], -acknowledgedAt? }`. `context` is the ANSI-stripped visible pane frame; + sessionName, kind: 'permission'|'question'|'idle', createdAt, toolName?, + toolSummary?, message?, cwd?, context?, options?: {n, label}[], + acknowledgedAt? }`. `context` is the ANSI-stripped visible pane frame; `options` is present only when the dialog's numbered choices parsed confidently; `acknowledgedAt` marks an item a human has already looked at (see `/viewed` below) and tells clients not to re-arm its tab alert. Listing @@ -474,7 +466,7 @@ acknowledgedAt? }`. `context` is the ANSI-stripped visible pane frame; first, `422 OPERATION_FAILED` when the session refused input. - `POST /api/v1/approvals/:id/dismiss` removes the item without keystrokes. - `POST /api/v1/approvals/session/:sessionId/viewed` → `{ sessionId, -acknowledged: itemId | null }`. Marks the session's pending **idle** item as + acknowledged: itemId | null }`. Marks the session's pending **idle** item as seen by a human (the web UI calls it when you open the session's tab): the item stays pending and answerable, but stops arming the yellow tab alert on every client, including after a reload. Permission/question items are never @@ -498,19 +490,19 @@ heuristic decides whether to ASK, never whether to act. Claude-mode sessions only (others carry their conversation id in their own config object); remote and docker sessions are never offered, because both need another host or container to be up. The plan is in-memory, so a server restart -drops it and the offer is gone — the conversations themselves are unaffected, +drops it and the offer is gone; the conversations themselves are unaffected, since they live in the CLI's own transcript store and stay reachable from the Resume list. A plan nobody spends expires after 24 hours. - `GET /api/v1/reboot-restore` → `{ sessions: RestorableSession[], -scrollbackRestored: false }`, ownership-scoped in multi-user mode. + scrollbackRestored: false }`, ownership-scoped in multi-user mode. `RestorableSession`: `{ id, name?, workingDir, mode, owner? }`. The persisted record itself is never sent. `scrollbackRestored` is always `false` and exists so a client states it: a restored session is a NEW pane, so the conversation continues and the terminal history does not. - `POST /api/v1/reboot-restore/restore` with `{ sessionIds?: string[] }` (omit to restore everything the caller can see) → `{ restored: RestorableSession[], -skipped: { sessionId, reason }[] }`. `reason` is one of `workspace-missing` + skipped: { sessionId, reason }[] }`. `reason` is one of `workspace-missing` (the directory is gone), `workspace-forbidden` (in multi-user mode it is outside the workspace of the user the session belongs to, re-checked against that owner's current grant rather than the caller's), `already-live` (the conversation is already @@ -521,7 +513,7 @@ skipped: { sessionId, reason }[] }`. `reason` is one of `workspace-missing` removed from the plan before any pane is built, so a double-click cannot put two panes on one conversation; anything that never became a pane goes back on offer, except `already-live`, which cannot stop being true. A restored session - comes back attached, idle and disarmed — respawn controllers and Ralph loops + comes back attached, idle and disarmed: respawn controllers and Ralph loops are never re-armed automatically. - `POST /api/v1/reboot-restore/dismiss` → `{ dismissed: n }`. Drops the offer for everything the caller can see. @@ -541,7 +533,7 @@ user guide: [`readmymind.md`](readmymind.md). - `GET /api/v1/sessions/:id/intent` -> `{ intent: IntentProfile }` for the session's case. `IntentProfile`: `{ key, workingDir, updatedAt, goals, -recentPrompts: { ts, sessionId, text }[] }` (prompts oldest first, FIFO cap + recentPrompts: { ts, sessionId, text }[] }` (prompts oldest first, FIFO cap 50, each <= 500 chars). A case with nothing recorded answers an empty profile with `updatedAt: 0`; nothing is persisted by reads. - `PUT /api/v1/sessions/:id/intent` with `{ goals }` (<= 8192 chars, strict @@ -574,7 +566,7 @@ same speech-to-text service the CLI's own `/voice` mode uses. Gated on the synce [`claude-voice-plan.md`](claude-voice-plan.md). - `GET /api/v1/voice/status` -> `{ available, reason?, subscriptionType?, -expiresAt? }`. `reason` is `disabled` (setting off), `no-credentials` (nobody + expiresAt? }`. `reason` is `disabled` (setting off), `no-credentials` (nobody signed in to Claude Code on the server), `expired` (the access token elapsed; running any Claude session refreshes it) or `malformed`. The OAuth token itself is never returned by this or any other endpoint. diff --git a/src/reboot-restore.ts b/src/reboot-restore.ts index 85760e8d..40b29b08 100644 --- a/src/reboot-restore.ts +++ b/src/reboot-restore.ts @@ -72,11 +72,17 @@ export interface RebootEvidence { * This heuristic decides whether to ASK, never whether to act. A wrong yes costs * the user a banner they dismiss, because the restore itself waits for a click. * - * ⚠️ `os.uptime()` reports the HOST's uptime, which a container shares. A Codeman - * running in Docker therefore sees a long uptime after its own container restarts, - * the boot test fails, and no banner appears. The feature is effectively off for - * containerized installs. That is the safe direction to fail in, and fixing it - * needs a boot signal the container actually owns rather than a wider heuristic. + * ⚠️ `os.uptime()` reports the HOST's uptime, which a container shares, and that + * cuts BOTH ways rather than simply switching the feature off in Docker. After a + * genuine host reboot a containerized Codeman sees the host's short uptime, so the + * banner DOES appear and the feature works. What it cannot see is a container-only + * restart: the host uptime is long, the boot test fails, and no banner appears + * although every in-container pane is gone (`docker/server.Dockerfile` installs + * tmux inside the Codeman container, and the self-updater restarts the Compose + * deployment by exiting the container, so that is the case where this would help + * most). Failing quiet is the safe direction, and closing the gap needs a boot + * signal the container owns (PID 1's start time, gated on the existing + * `isRunningInContainer()`) rather than a wider heuristic. */ export function looksLikeHostReboot(evidence: RebootEvidence): boolean { if (evidence.deadSessionCount === 0) return false; diff --git a/src/web/ports/session-port.ts b/src/web/ports/session-port.ts index 85add1d8..78e1fbbd 100644 --- a/src/web/ports/session-port.ts +++ b/src/web/ports/session-port.ts @@ -27,7 +27,19 @@ export interface SessionPort { reapplyPersistedSessionState( session: Session, saved: SessionState, - phase: 'before-spawn' | 'after-spawn' + phase: 'before-spawn' | 'after-spawn', + options?: { + /** + * Re-arm a PENDING auto-resume schedule from the record's `autoResumeAt`. + * Default true, which is what a Codeman restart wants: the limit footer + * will not reprint on its own, so dropping the stamp there strands the + * pause. A reboot restore passes false: the stamp predates the reboot, + * the pane is new, and re-arming means every restored session types + * `continue` into itself about a minute after one click. Auto-resume + * stays ENABLED either way, so it re-arms on fresh evidence. + */ + rearmAutoResumeSchedule?: boolean; + } ): Promise<void>; /** * Undo a session that was registered but never got a working pane: the map diff --git a/src/web/public/reboot-restore-ui.js b/src/web/public/reboot-restore-ui.js index 2a3599dd..1e738bdd 100644 --- a/src/web/public/reboot-restore-ui.js +++ b/src/web/public/reboot-restore-ui.js @@ -26,7 +26,7 @@ * @mixin Extends CodemanApp.prototype via Object.assign * @dependency app.js (CodemanApp class, showToast) * @dependency api-client.js at runtime (this._api / this._apiJson) - * @loadorder 11.7 of 17, after approvals-ui.js + * @loadorder 11.65, after approvals-ui.js and before admin-ui.js (11.7) */ /** Plain-language wording for one skip reason, for the toast after a restore. */ diff --git a/src/web/public/styles.css b/src/web/public/styles.css index f35bf5cd..975b4d8b 100644 --- a/src/web/public/styles.css +++ b/src/web/public/styles.css @@ -2548,6 +2548,10 @@ body.solo-mode .header-tokens, body.solo-mode .btn-notifications, body.solo-mode .btn-multimonitor, body.solo-mode .header-plan-usage, +/* A solo window shows ONE session and has no tab strip to put restored ones in, + so offering to rebuild a list of them there is an offer it cannot show the + result of. The dashboard that spawned this window carries the banner. */ +body.solo-mode .reboot-restore-banner, body.solo-mode .btn-lifecycle-log { display: none !important; } diff --git a/src/web/routes/reboot-restore-routes.ts b/src/web/routes/reboot-restore-routes.ts index 3ff27b4f..fb6cec97 100644 --- a/src/web/routes/reboot-restore-routes.ts +++ b/src/web/routes/reboot-restore-routes.ts @@ -41,7 +41,7 @@ import { clampEnvOverridesForOwner } from '../../session-env-clamp.js'; import { Session } from '../../session.js'; import { resolveClaudeModeForUsername } from '../../user-store.js'; import { getCli } from '../../config/cli-registry/registry.js'; -import { applyWorkspaceHooks } from '../../hooks-config.js'; +import { applyWorkspaceHooks, seedAgentSessionPreamble } from '../../hooks-config.js'; import { getLifecycleLog } from '../../session-lifecycle-log.js'; import { STATS_COLLECTION_INTERVAL_MS } from '../../config/server-timing.js'; import { SseEvent } from '../sse-events.js'; @@ -105,11 +105,16 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes // the user resumed by hand from the Resume list is already on screen, and a // second pane on it would fight the first for the same transcript. This one // is never re-offered: unlike a missing workspace, it cannot stop being true. - const liveSessionIds = new Set(ctx.sessions.keys()); - const liveConversationIds = new Set( - [...ctx.sessions.values()].map((session) => session.claudeSessionId).filter((id): id is string => !!id) - ); - const { restore, skipped } = rejectAlreadyLive(taken, liveSessionIds, liveConversationIds); + // Read fresh each time rather than snapshotted once: the loop below awaits a + // real `startInteractive()` per entry, so by the tenth entry a snapshot taken + // here is tens of seconds old, and a conversation the user resumed by hand in + // that window would be invisible to it. + const liveSessionIds = () => new Set(ctx.sessions.keys()); + const liveConversationIds = () => + new Set( + [...ctx.sessions.values()].map((session) => session.claudeSessionId).filter((id): id is string => !!id) + ); + const { restore, skipped } = rejectAlreadyLive(taken, liveSessionIds(), liveConversationIds()); for (const entry of taken) { if (skipped.some((s) => s.sessionId === entry.sessionId)) unspent.delete(entry); } @@ -119,6 +124,18 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes const workspaceHooksEnabled = await ctx.getWorkspaceHooksEnabled(); for (const entry of restore) { + // The already-live check, re-run against the board as it is NOW. The pass + // above decided the batch; this catches a conversation that went live while + // an earlier entry in this same batch was starting. Spent rather than + // returned to the plan, for the same reason as the batch pass: unlike a + // missing workspace or a withdrawn grant, an open conversation is not a + // condition that stops being true. + const [lateLive] = rejectAlreadyLive([entry], liveSessionIds(), liveConversationIds()).skipped; + if (lateLive) { + failures.push(lateLive); + unspent.delete(entry); + continue; + } // Capacity is re-checked per iteration, because this loop is itself // creating the sessions it counts. The offer can be a day old, so the // board may be fuller now than the plan assumed. @@ -157,6 +174,12 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes workingDir: saved.workingDir, mode: saved.mode, name: saved.name, + // Without this the constructor re-infers ownership from the name, so a + // session the user renamed by hand to something shaped like `w<n>-<case>` + // comes back as `placeholder` and auto-naming overwrites their name on + // the next prompt. The route persists below, so the loss would go to + // disk. `restoreMuxSessions()` passes it for the same reason. + nameSource: saved.nameSource, createdAt: saved.createdAt, mux: ctx.mux, useMux: true, @@ -197,7 +220,14 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes // the reduced one and drop the pin that keeps it from being pruned. A // listener-driven persist can still land inside the debounce window // while the pane starts; the write below repairs the record. - await ctx.reapplyPersistedSessionState(session, saved, 'after-spawn'); + // `rearmAutoResumeSchedule: false`: the saved stamp predates the reboot and + // the pane is new, so honouring it would have every restored session type + // `continue` into itself about a minute after one click. Auto-resume stays + // enabled and re-arms on the next real limit message. This is also what the + // module header promises ("comes back attached, idle and disarmed"). + await ctx.reapplyPersistedSessionState(session, saved, 'after-spawn', { + rearmAutoResumeSchedule: false, + }); ctx.persistSessionState(session); // A session without its workspace hooks goes silently blind: no stop or @@ -211,6 +241,16 @@ export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRes ); } + // Both create paths seed this; without it a restored claude session's agent + // skill falls back to writing out the whole ~150-line §0 preamble. Remote and + // docker sessions never reach here (the plan rejects them as + // `remote-or-docker`), so the local-only condition is structural. + if (getCli(session.mode)?.capabilities.agentSkillInjection && (await ctx.getAgentSkillEnabled())) { + await seedAgentSessionPreamble(session.id).catch((err: unknown) => + console.warn(`[agent-skill] preamble seed failed for ${session.id}: ${getErrorMessage(err)}`) + ); + } + getLifecycleLog().log({ event: 'recovered', sessionId: session.id, name: session.name }); // Every other open tab and phone needs this; the clicking tab already has // the response, and the client's handler is an idempotent upsert. diff --git a/src/web/server.ts b/src/web/server.ts index e94b8cdf..3497afdf 100644 --- a/src/web/server.ts +++ b/src/web/server.ts @@ -2940,7 +2940,8 @@ export class WebServer extends EventEmitter { async reapplyPersistedSessionState( session: Session, saved: SessionState, - phase: 'before-spawn' | 'after-spawn' + phase: 'before-spawn' | 'after-spawn', + options?: { rearmAutoResumeSchedule?: boolean } ): Promise<void> { if (phase === 'before-spawn') { // The custom-model env has to be rebuilt from the endpoint store: the persist @@ -2968,7 +2969,14 @@ export class WebServer extends EventEmitter { session.setAutoClear(saved.autoClearEnabled ?? false, saved.autoClearThreshold); } if (saved.autoResumeEnabled) { - session.restoreAutoResume(true, saved.autoResumeAt); + // The stamp is re-armed by default, because a Codeman restart leaves the + // limit footer un-reprinted and dropping it there would strand the pause. + // A reboot restore opts out: that stamp predates the reboot, the pane is + // new, and honouring it means every session the user restored types + // `continue` into itself about a minute later, unattended. The setting + // itself stays on either way, so it re-arms on the next limit message. + const rearm = options?.rearmAutoResumeSchedule !== false; + session.restoreAutoResume(true, rearm ? saved.autoResumeAt : undefined); } if (saved.inputTokens !== undefined || saved.outputTokens !== undefined || saved.totalCost !== undefined) { session.restoreTokens(saved.inputTokens ?? 0, saved.outputTokens ?? 0, saved.totalCost ?? 0); @@ -3024,9 +3032,18 @@ export class WebServer extends EventEmitter { session.ralphTracker.stopWatchingFixPlan(); const summaryTracker = this.runSummaryTrackers.get(sessionId); if (summaryTracker) { + // Closes the run's own record before the tracker goes, the way + // `_doCleanupSession()` does. Cosmetic rather than load-bearing, but a + // run left open reads as still going in the away digest. + summaryTracker.recordSessionStopped(); summaryTracker.stop(); this.runSummaryTrackers.delete(sessionId); } + // Also mirrors `_doCleanupSession()`. The PERSISTED Ralph state is left + // alone on purpose (that is one of the things separating this from + // cleanupSession); this only clears the in-memory tracker the failed + // construction built, which the retry reuses the id of. + session.ralphTracker.fullReset(); // --- what anything else may have attached to this id in the meantime --- // A rebuild can fail AFTER startInteractive() resolved, and a restored diff --git a/test/routes/reboot-restore-routes.test.ts b/test/routes/reboot-restore-routes.test.ts index dcb7d7ed..19067d21 100644 --- a/test/routes/reboot-restore-routes.test.ts +++ b/test/routes/reboot-restore-routes.test.ts @@ -11,7 +11,10 @@ * The routes read the process-wide `rebootRestoreRegistry` singleton, so every * test resets it; a leaked entry would bleed into the next one. */ -import { describe, it, expect, afterEach } from 'vitest'; +import { describe, it, expect, afterEach, beforeEach } from 'vitest'; +import { mkdtempSync, rmSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; import Fastify, { type FastifyInstance } from 'fastify'; import fastifyCookie from '@fastify/cookie'; import { registerRebootRestoreRoutes } from '../../src/web/routes/reboot-restore-routes.js'; @@ -187,6 +190,49 @@ describe('POST /api/reboot-restore/restore', () => { }); }); +describe('POST /api/reboot-restore/restore: multi-user workspace confinement', () => { + const saved: Record<string, string | undefined> = {}; + let realDir: string; + + beforeEach(() => { + saved.CODEMAN_MULTIUSER = process.env.CODEMAN_MULTIUSER; + process.env.CODEMAN_MULTIUSER = '1'; + // This branch sits AFTER the existsSync check, so the workspace has to be + // real for the confinement rule to be the thing that rejects the entry. + realDir = mkdtempSync(join(tmpdir(), 'codeman-reboot-restore-real-')); + }); + + afterEach(() => { + if (saved.CODEMAN_MULTIUSER === undefined) delete process.env.CODEMAN_MULTIUSER; + else process.env.CODEMAN_MULTIUSER = saved.CODEMAN_MULTIUSER; + rmSync(realDir, { recursive: true, force: true }); + }); + + it("refuses a workspace outside the OWNER's case space, and leaves it on offer", async () => { + const entry = offerEntry('a', 'alice'); + entry.workingDir = realDir; + (entry.state as { workingDir: string }).workingDir = realDir; + rebootRestoreRegistry.set([entry]); + + // An admin does the clicking. The confinement is still resolved against + // alice, the entry's OWNER: `isWorkingDirAllowed` waves an admin through, so + // reading the caller here would hand an admin the power to rebuild another + // user's session anywhere on the box. + const app = await createHarness({ username: 'root-user', role: 'admin' }); + const res = await app.inject({ method: 'POST', url: '/api/reboot-restore/restore', payload: {} }); + + expect(res.statusCode).toBe(200); + expect(res.json().data.restored).toEqual([]); + expect(res.json().data.skipped).toEqual([{ sessionId: 'a', reason: 'workspace-forbidden' }]); + + // A withdrawn grant can be given back, so unlike `already-live` this is not + // the permanent kind of refusal and the entry stays claimable. + const left = (await app.inject({ method: 'GET', url: '/api/reboot-restore' })).json().data; + expect(left.sessions.map((s: { id: string }) => s.id)).toEqual(['a']); + await app.close(); + }); +}); + describe('POST /api/reboot-restore/dismiss', () => { it('drops the offer and leaves the banner with nothing to show', async () => { rebootRestoreRegistry.set([offerEntry('a'), offerEntry('b')]); From 0e1191b774ad7f1ac32696ee0a0d9de1d3496904 Mon Sep 17 00:00:00 2001 From: Codeman maintainer <noreply@anthropic.com> Date: Fri, 18 Sep 2026 13:55:22 +0200 Subject: [PATCH 23/28] chore(changeset): trim the contributor entries and add the 1.30.0 thanks Changeset text becomes user-facing CHANGELOG, so the #429 entry is cut from five bullets of internal bash-array detail down to what the change does for someone running the installer, as promised on the PR. The #441 entry loses its em-dashes, which are not house style. Adds an entry for the maintainer fixes applied while landing #442, and the Thanks section. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- .changeset/android-last-character-on-enter.md | 2 +- .changeset/cli-catalog-followups.md | 21 +------------------ .changeset/release-1-30-0-merge-fixes.md | 5 +++++ .changeset/release-1-30-0-thanks.md | 9 ++++++++ 4 files changed, 16 insertions(+), 21 deletions(-) create mode 100644 .changeset/release-1-30-0-merge-fixes.md create mode 100644 .changeset/release-1-30-0-thanks.md diff --git a/.changeset/android-last-character-on-enter.md b/.changeset/android-last-character-on-enter.md index 0af4c14a..c68df755 100644 --- a/.changeset/android-last-character-on-enter.md +++ b/.changeset/android-last-character-on-enter.md @@ -2,4 +2,4 @@ "aicodeman": patch --- -Stop a phone keyboard losing the last character of every message it sends. Android soft keyboards commit the last typed character and send the Enter key in one InputConnection transaction, so the `input` event and the Enter keydown are both processed before any zero-delay timer runs. The orphaned-input recovery from #388 only resolved its candidate on such a timer, and lost it both ways: xterm emits `\r` synchronously from the Enter keydown, so the local-echo composer submitted the prompt before the recovered character existed, and that `\r` bumped the "did xterm speak for this keystroke" counter, so the candidate then stood itself down and dropped the character outright. Pending candidates are now drained synchronously at the next keydown, from xterm's custom key handler — before xterm processes that key — so the counter still holds the value it had while the candidate's own keystroke was current, and the recovered byte reaches the composer ahead of the Enter. Typing on a physical keyboard is unaffected: there, the timer has already resolved the candidate before the next key arrives. +Stop a phone keyboard losing the last character of every message it sends. Android soft keyboards commit the last typed character and send the Enter key in one InputConnection transaction, so the `input` event and the Enter keydown are both processed before any zero-delay timer runs. The orphaned-input recovery from #388 only resolved its candidate on such a timer, and lost it both ways: xterm emits `\r` synchronously from the Enter keydown, so the local-echo composer submitted the prompt before the recovered character existed, and that `\r` bumped the "did xterm speak for this keystroke" counter, so the candidate then stood itself down and dropped the character outright. Pending candidates are now drained synchronously at the next keydown, from xterm's custom key handler, which runs before xterm processes that key, so the counter still holds the value it had while the candidate's own keystroke was current, and the recovered byte reaches the composer ahead of the Enter. Typing on a physical keyboard is unaffected: there, the timer has already resolved the candidate before the next key arrives. diff --git a/.changeset/cli-catalog-followups.md b/.changeset/cli-catalog-followups.md index 5970e119..1e16cbcf 100644 --- a/.changeset/cli-catalog-followups.md +++ b/.changeset/cli-catalog-followups.md @@ -2,23 +2,4 @@ "aicodeman": patch --- -Cleans up the loose ends the maintainer flagged as "worth knowing rather than fixing" when -merging the CLI-catalogue-driven `install.sh`/Docker-agent-image PR (#380): - -- `install.sh` no longer carries `_cli_index`/`check_cli`/`get_cli_path`, three generic - lookup helpers left behind once the catalogue-driven menu and hints stopped calling them. -- The generator no longer emits `CLI_KIND`/`CLI_NPM`, two bash arrays nothing in `install.sh` - read (the `.mjs`/`docker-hosts.ts` producers already read the JSON catalogue's `kind`/ - `npmPackage` fields directly). -- `detect_all_clis` now skips a disabled entry entirely rather than probing it and filtering - the result downstream — no stock entry ships disabled today, so this is a latent - inefficiency closed before it is a latent bug, not a behaviour change. -- The install hint for a `launcherProfile` entry (DeepSeek today) now explains, in one line, - why it is a docs link rather than a runnable command — its docs page documents - `npm install -g @deepseek-ai/dsh`, which installs the launcher only and cannot drive a pane - on its own, the exact trap the menu already avoids by withholding the command. Driven by a - new generated `CLI_LAUNCHER_ONLY` array (from `discovery.launcherProfile`), not an id check. -- The non-interactive default's comment no longer claims it is always Claude Code: on a - wget-only host, Claude's curl one-liner is filtered out of the offered list first, so the - default becomes whichever npm-based entry sorts earliest instead. Behaviour is unchanged - (and was already printed, so never silent) — only the comment was wrong. +The installer's hint for a launcher-only CLI (DeepSeek today) now says why it is a docs link rather than a command you can run, and points at the thing that resolves it: the package installs a launcher that still needs a terminal profile, and Codeman's Run menu can add one in a click. Driven by a generated `CLI_LAUNCHER_ONLY` flag rather than an id check, so it covers any future entry of that shape. Also removes three dead lookup helpers and two never-read generated arrays from `install.sh`, skips a disabled entry's probe instead of filtering it afterwards, and corrects a comment that claimed the non-interactive default is always Claude Code (on a wget-only host its curl one-liner is filtered out first). diff --git a/.changeset/release-1-30-0-merge-fixes.md b/.changeset/release-1-30-0-merge-fixes.md new file mode 100644 index 00000000..71f6e65b --- /dev/null +++ b/.changeset/release-1-30-0-merge-fixes.md @@ -0,0 +1,5 @@ +--- +"aicodeman": patch +--- + +Maintainer fixes applied while landing the above. A session restored after a reboot keeps the name you gave it (the rebuild dropped the field that records who named a session, so a hand-renamed session came back looking auto-named and the next prompt overwrote it), and no longer types `continue` into itself on its own: a pending auto-resume stamp from before the reboot is dropped rather than re-armed, since the pane is new and one click could otherwise arm several unattended prompts at once. Auto-resume itself stays on and re-arms on the next real usage-limit message. The restore offer is also hidden in a detached single-session window, which has no tab strip to put restored sessions in, and a conversation that goes live while an earlier session in the same batch is starting is no longer restored a second time. diff --git a/.changeset/release-1-30-0-thanks.md b/.changeset/release-1-30-0-thanks.md new file mode 100644 index 00000000..767cad6c --- /dev/null +++ b/.changeset/release-1-30-0-thanks.md @@ -0,0 +1,9 @@ +--- +"aicodeman": patch +--- + +### Thanks + +- @irisitymichaelgrundberg for the reboot-restore banner (#442), and for the three real reboots behind it rather than a mocked one. +- @shenlvkang-collab for tracking down why Android keyboards lost the last character of every message (#441), including the half where the character was not late but gone. +- @opticon454 for going back and closing out the loose ends left as "worth knowing rather than fixing" after #380 (#429). From 20fc7b3c3d25bc5646f1dda0f02b20c1d40e1bd0 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 14:09:55 +0200 Subject: [PATCH 24/28] chore: version packages (#447) * chore: version packages * chore: sync the CLAUDE.md version line to 1.30.0 The changesets bot does not touch this line, and pushing it to master after merging the version PR starts a second Release run that has raced the first before. Riding the bot's own branch keeps it to one push. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --------- Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: Codeman maintainer <noreply@anthropic.com> --- .changeset/android-last-character-on-enter.md | 5 ----- .changeset/cli-catalog-followups.md | 5 ----- .changeset/reboot-restore-banner.md | 5 ----- .changeset/release-1-30-0-merge-fixes.md | 5 ----- .changeset/release-1-30-0-thanks.md | 9 --------- .../terminal-history-anchor-after-parse.md | 5 ----- .claude-plugin/marketplace.json | 2 +- CHANGELOG.md | 18 ++++++++++++++++++ CLAUDE.md | 2 +- package-lock.json | 4 ++-- package.json | 2 +- plugins/codeman/.claude-plugin/plugin.json | 2 +- 12 files changed, 24 insertions(+), 40 deletions(-) delete mode 100644 .changeset/android-last-character-on-enter.md delete mode 100644 .changeset/cli-catalog-followups.md delete mode 100644 .changeset/reboot-restore-banner.md delete mode 100644 .changeset/release-1-30-0-merge-fixes.md delete mode 100644 .changeset/release-1-30-0-thanks.md delete mode 100644 .changeset/terminal-history-anchor-after-parse.md diff --git a/.changeset/android-last-character-on-enter.md b/.changeset/android-last-character-on-enter.md deleted file mode 100644 index c68df755..00000000 --- a/.changeset/android-last-character-on-enter.md +++ /dev/null @@ -1,5 +0,0 @@ ---- -"aicodeman": patch ---- - -Stop a phone keyboard losing the last character of every message it sends. Android soft keyboards commit the last typed character and send the Enter key in one InputConnection transaction, so the `input` event and the Enter keydown are both processed before any zero-delay timer runs. The orphaned-input recovery from #388 only resolved its candidate on such a timer, and lost it both ways: xterm emits `\r` synchronously from the Enter keydown, so the local-echo composer submitted the prompt before the recovered character existed, and that `\r` bumped the "did xterm speak for this keystroke" counter, so the candidate then stood itself down and dropped the character outright. Pending candidates are now drained synchronously at the next keydown, from xterm's custom key handler, which runs before xterm processes that key, so the counter still holds the value it had while the candidate's own keystroke was current, and the recovered byte reaches the composer ahead of the Enter. Typing on a physical keyboard is unaffected: there, the timer has already resolved the candidate before the next key arrives. diff --git a/.changeset/cli-catalog-followups.md b/.changeset/cli-catalog-followups.md deleted file mode 100644 index 1e16cbcf..00000000 --- a/.changeset/cli-catalog-followups.md +++ /dev/null @@ -1,5 +0,0 @@ ---- -"aicodeman": patch ---- - -The installer's hint for a launcher-only CLI (DeepSeek today) now says why it is a docs link rather than a command you can run, and points at the thing that resolves it: the package installs a launcher that still needs a terminal profile, and Codeman's Run menu can add one in a click. Driven by a generated `CLI_LAUNCHER_ONLY` flag rather than an id check, so it covers any future entry of that shape. Also removes three dead lookup helpers and two never-read generated arrays from `install.sh`, skips a disabled entry's probe instead of filtering it afterwards, and corrects a comment that claimed the non-interactive default is always Claude Code (on a wget-only host its curl one-liner is filtered out first). diff --git a/.changeset/reboot-restore-banner.md b/.changeset/reboot-restore-banner.md deleted file mode 100644 index 33942048..00000000 --- a/.changeset/reboot-restore-banner.md +++ /dev/null @@ -1,5 +0,0 @@ ---- -'aicodeman': minor ---- - -Offer to rebuild the sessions a host reboot destroyed. A reboot takes the tmux server down with it, so every pane dies and the board comes up empty. Codeman now works out what was running, and the board offers to restore it behind a click. The conversations come back; the terminal scrollback does not, and the banner says so. diff --git a/.changeset/release-1-30-0-merge-fixes.md b/.changeset/release-1-30-0-merge-fixes.md deleted file mode 100644 index 71f6e65b..00000000 --- a/.changeset/release-1-30-0-merge-fixes.md +++ /dev/null @@ -1,5 +0,0 @@ ---- -"aicodeman": patch ---- - -Maintainer fixes applied while landing the above. A session restored after a reboot keeps the name you gave it (the rebuild dropped the field that records who named a session, so a hand-renamed session came back looking auto-named and the next prompt overwrote it), and no longer types `continue` into itself on its own: a pending auto-resume stamp from before the reboot is dropped rather than re-armed, since the pane is new and one click could otherwise arm several unattended prompts at once. Auto-resume itself stays on and re-arms on the next real usage-limit message. The restore offer is also hidden in a detached single-session window, which has no tab strip to put restored sessions in, and a conversation that goes live while an earlier session in the same batch is starting is no longer restored a second time. diff --git a/.changeset/release-1-30-0-thanks.md b/.changeset/release-1-30-0-thanks.md deleted file mode 100644 index 767cad6c..00000000 --- a/.changeset/release-1-30-0-thanks.md +++ /dev/null @@ -1,9 +0,0 @@ ---- -"aicodeman": patch ---- - -### Thanks - -- @irisitymichaelgrundberg for the reboot-restore banner (#442), and for the three real reboots behind it rather than a mocked one. -- @shenlvkang-collab for tracking down why Android keyboards lost the last character of every message (#441), including the half where the character was not late but gone. -- @opticon454 for going back and closing out the loose ends left as "worth knowing rather than fixing" after #380 (#429). diff --git a/.changeset/terminal-history-anchor-after-parse.md b/.changeset/terminal-history-anchor-after-parse.md deleted file mode 100644 index 418171c4..00000000 --- a/.changeset/terminal-history-anchor-after-parse.md +++ /dev/null @@ -1,5 +0,0 @@ ---- -"aicodeman": patch ---- - -Keep the terminal anchored where you are reading while an agent streams (#358). Scrolling up during a Codex response could still be dragged back to the live bottom by the next redraw: the flush captured the viewport before writing and restored it immediately after, but xterm parses asynchronously, so at that moment the buffer had not moved yet, the restore compared the anchor against itself and did nothing, and the redraw landed a tick later with nothing left to pull the view back. The restore now runs inside xterm's own write callback, which is the first point at which the redraw's effect exists, and it holds across consecutive and chunked redraws. It is dropped if you switch sessions or a history replay starts before the write parses, since the anchor indexes the buffer it was captured from. diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 1d5433b4..7402d884 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -10,7 +10,7 @@ "name": "codeman", "source": "./plugins/codeman", "description": "Drive Codeman from inside a Claude Code session: spawn worker sessions, prompt them, wait for them, read their answers, clean up. Acts only inside a Codeman-managed session.", - "version": "1.29.1", + "version": "1.30.0", "author": { "name": "Ark0N", "url": "https://github.com/Ark0N" diff --git a/CHANGELOG.md b/CHANGELOG.md index c2740d9e..10fdcd21 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,23 @@ # aicodeman +## 1.30.0 + +### Minor Changes + +- da933d7: Offer to rebuild the sessions a host reboot destroyed. A reboot takes the tmux server down with it, so every pane dies and the board comes up empty. Codeman now works out what was running, and the board offers to restore it behind a click. The conversations come back; the terminal scrollback does not, and the banner says so. + +### Patch Changes + +- a1c35da: Stop a phone keyboard losing the last character of every message it sends. Android soft keyboards commit the last typed character and send the Enter key in one InputConnection transaction, so the `input` event and the Enter keydown are both processed before any zero-delay timer runs. The orphaned-input recovery from #388 only resolved its candidate on such a timer, and lost it both ways: xterm emits `\r` synchronously from the Enter keydown, so the local-echo composer submitted the prompt before the recovered character existed, and that `\r` bumped the "did xterm speak for this keystroke" counter, so the candidate then stood itself down and dropped the character outright. Pending candidates are now drained synchronously at the next keydown, from xterm's custom key handler, which runs before xterm processes that key, so the counter still holds the value it had while the candidate's own keystroke was current, and the recovered byte reaches the composer ahead of the Enter. Typing on a physical keyboard is unaffected: there, the timer has already resolved the candidate before the next key arrives. +- 3f2928a: The installer's hint for a launcher-only CLI (DeepSeek today) now says why it is a docs link rather than a command you can run, and points at the thing that resolves it: the package installs a launcher that still needs a terminal profile, and Codeman's Run menu can add one in a click. Driven by a generated `CLI_LAUNCHER_ONLY` flag rather than an id check, so it covers any future entry of that shape. Also removes three dead lookup helpers and two never-read generated arrays from `install.sh`, skips a disabled entry's probe instead of filtering it afterwards, and corrects a comment that claimed the non-interactive default is always Claude Code (on a wget-only host its curl one-liner is filtered out first). +- 0e1191b: Maintainer fixes applied while landing the above. A session restored after a reboot keeps the name you gave it (the rebuild dropped the field that records who named a session, so a hand-renamed session came back looking auto-named and the next prompt overwrote it), and no longer types `continue` into itself on its own: a pending auto-resume stamp from before the reboot is dropped rather than re-armed, since the pane is new and one click could otherwise arm several unattended prompts at once. Auto-resume itself stays on and re-arms on the next real usage-limit message. The restore offer is also hidden in a detached single-session window, which has no tab strip to put restored sessions in, and a conversation that goes live while an earlier session in the same batch is starting is no longer restored a second time. +- 0e1191b: ### Thanks + - @irisitymichaelgrundberg for the reboot-restore banner (#442), and for the three real reboots behind it rather than a mocked one. + - @shenlvkang-collab for tracking down why Android keyboards lost the last character of every message (#441), including the half where the character was not late but gone. + - @opticon454 for going back and closing out the loose ends left as "worth knowing rather than fixing" after #380 (#429). + +- de864e7: Keep the terminal anchored where you are reading while an agent streams (#358). Scrolling up during a Codex response could still be dragged back to the live bottom by the next redraw: the flush captured the viewport before writing and restored it immediately after, but xterm parses asynchronously, so at that moment the buffer had not moved yet, the restore compared the anchor against itself and did nothing, and the redraw landed a tick later with nothing left to pull the view back. The restore now runs inside xterm's own write callback, which is the first point at which the redraw's effect exists, and it holds across consecutive and chunked redraws. It is dropped if you switch sessions or a history replay starts before the write parses, since the anchor indexes the buffer it was captured from. + ## 1.29.1 ### Patch Changes diff --git a/CLAUDE.md b/CLAUDE.md index b3d765ac..196ad22e 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -77,7 +77,7 @@ When user says "COM": CI runs `npm run check:lockfile` on every push/PR, so lockfile drift fails the build even if the `version-packages` script is bypassed. -**Version**: 1.29.1 (must match `package.json`) +**Version**: 1.30.0 (must match `package.json`) ## Project Overview diff --git a/package-lock.json b/package-lock.json index 22edc788..f9640246 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "aicodeman", - "version": "1.29.1", + "version": "1.30.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "aicodeman", - "version": "1.29.1", + "version": "1.30.0", "hasInstallScript": true, "license": "MIT", "workspaces": [ diff --git a/package.json b/package.json index 287879ba..48a185f6 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "aicodeman", - "version": "1.29.1", + "version": "1.30.0", "description": "Mission control for AI coding agents - run 20 autonomous agents with real-time monitoring and session persistence", "type": "module", "main": "dist/index.js", diff --git a/plugins/codeman/.claude-plugin/plugin.json b/plugins/codeman/.claude-plugin/plugin.json index 8d16d73b..c2def406 100644 --- a/plugins/codeman/.claude-plugin/plugin.json +++ b/plugins/codeman/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "codeman", "description": "Drive Codeman, the self-hosted session manager for AI coding agents, from inside a Claude Code session: spawn worker sessions, prompt them, wait for them, read their answers, clean up. Acts only inside a Codeman-managed session.", - "version": "1.29.1", + "version": "1.30.0", "author": { "name": "Ark0N", "url": "https://github.com/Ark0N" From 75a028e825d8e38dc55bf42c4b618c92566df114 Mon Sep 17 00:00:00 2001 From: Michael Grundberg <michael.grundberg@irisity.com> Date: Fri, 18 Sep 2026 16:43:04 +0200 Subject: [PATCH 25/28] fix(terminal): re-take the sticky-scroll baseline after a replay MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A capture load now replays its queued tail, and that replay runs through `batchTerminalWrite`, which samples `_wasAtBottomBeforeWrite` before it queues. It runs inside `chunkedTerminalWrite`, before that promise resolves, with the terminal freshly reset and rewritten — so the sample is always true. The caller then restored the reader's position and the next `flushPendingWrites` scrolled straight back to the bottom off the latched flag, undoing it. The only thing in the way was `_hasRecentUserScrollUp()`, a 1500ms window a server-triggered refresh is usually past. `_syncStickyScrollBaseline()` re-takes the flag from wherever the viewport now sits, and the two paths that restore a position call it right after doing so: `_onSessionNeedsRefresh` and `_maybeRefetchFullHistory`. Those are the paths #259 and #205 exist for, and they are also where a non-empty queue is most likely, since a needsRefresh fires when output is flooding. Re-taking rather than suppressing the sampling: suppressing leaves whatever stale value the flag held from before the load, which on the full-history re-pull has no reason to be false. `selectSession` and `_onSessionClearTerminal` deliberately end at the bottom, so the sampled true is already the truth there and they do not call it. `_bufferLoadFinishOpts` gains the coverage the CI gate can see: both mux sources flush, `history` does not, and a payload naming no source does not. Its only coverage was the browser suite, which CI does not run. The JSDoc and the changeset now record the one duplicate window this cutoff cannot close. The server appends output to the byte buffer in the same tick it emits, but broadcasts on a batch timer — 8ms over WebSocket, 16 to 50ms over SSE — so a batch pending when `capture-pane` ran leaves the server after the reply and is replayed although the capture holds it. It is one batch interval wide against a recovery window spanning the whole chunked write, and closing it means flushing that batch server side before the capture. The second browser test asserts its session was created, so a failed create fails it instead of passing with zero hits. docs/architecture-invariants.md no longer claims the replay leaves the queued-event discard window alone. That clause now describes what decides how a load ends, the baseline rule, the batch window, and the three covering tests. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- ...y-output-that-arrived-after-the-capture.md | 15 +++ docs/architecture-invariants.md | 2 +- src/web/public/app.js | 20 ++++ src/web/public/terminal-ui.js | 21 ++++ test/capture-load-window.browser.test.ts | 3 + test/terminal-buffer-flush.test.ts | 98 +++++++++++++++++++ test/terminal-flush-budget.test.ts | 45 +++++++++ 7 files changed, 203 insertions(+), 1 deletion(-) diff --git a/.changeset/fix-replay-output-that-arrived-after-the-capture.md b/.changeset/fix-replay-output-that-arrived-after-the-capture.md index 03e09189..0cb24af0 100644 --- a/.changeset/fix-replay-output-that-arrived-after-the-capture.md +++ b/.changeset/fix-replay-output-that-arrived-after-the-capture.md @@ -34,3 +34,18 @@ takes the flush policy and applies it at its own finish sites. `_beginBufferLoad` no longer empties the queue when the same load re-enters it, which it does on every write, because that reset discarded the fetch window before anything could replay it. + +A path that replays its queue and then restores a scroll position re-takes the +sticky-scroll baseline (`_syncStickyScrollBaseline`). The replay runs with the +terminal freshly reset, so it reads as sitting at the bottom, and the next flush +would scroll there and undo the restore. The backpressure refresh and the +full-history re-pull are the two paths that restore a position, and both are +ones a reader reaches while scrolled up. + +One duplicate window stays open and is not closable from the browser. The server +appends output to the byte buffer in the same tick it emits, but broadcasts on a +batch timer, 8ms over WebSocket and 16 to 50ms over SSE. A batch already pending +when `capture-pane` ran therefore leaves the server after the reply and is +replayed although the capture holds it. It is one batch interval wide, against a +recovery window that spans the whole chunked write, and closing it means +flushing that session's pending batch before taking the capture. diff --git a/docs/architecture-invariants.md b/docs/architecture-invariants.md index 20f6bda9..3a0e8733 100644 --- a/docs/architecture-invariants.md +++ b/docs/architecture-invariants.md @@ -116,7 +116,7 @@ Tests: `test/docker-hosts.test.ts`, `test/docker-exec-options.test.ts`, `test/do ### Full-scrollback replay -**Full-scrollback replay** (COD-164/#148, reworked for #205): `GET /api/sessions/:id/terminal?full=1` returns the ENTIRE tmux scrollback (capture-pane `-e -S -<lines>` bounded by the configured history limit, explicit `maxBuffer` from the terminal-history config, early byte-cap before normalization, CRLF-normalized for shell panes). On success the capture is returned ALONE (`source='mux-full-history'` — it supersedes the byte buffer; no duplication). The first load of each non-shell TUI session per page requests `full=1` (`_fullHistoryLoaded` Set in app.js — the old one-shot `_initialFullBufferLoad` flag was consumed by whichever tab auto-selected, leaving every other TUI tab one frame of history). Shell sessions instead load a bounded 1 MiB `?tail=` window on every selection and automatic drop recovery: a 100k-line shell capture can be tens of MiB, and automatically parsing it makes tab-switch latency scale with the entire session. Shell full history is explicit-button-only; reaching the top during an ordinary wheel/touch gesture must not reset xterm and replay the multi-megabyte capture on its main thread. Other modes may still re-pull `full=1` at the TOP, and pressing **Load full history** forces the request for any recoverably truncated session (`_maybeRefetchFullHistory`, 4s per-session gesture cooldown, in-flight + tab-switch guards, viewport position held across the replay); Shell full pulls are not retained in the tab cache, so the next switch stays bounded. Chunked replay enqueues 32 KiB pieces across safe yields, appends an xterm parse marker, then releases the live-output gate; output arriving after that release stays ordered behind the snapshot, while the marker callback supplies accurate parse timing without extending the pre-existing queued-event discard window. Live output is separately one-chunk-in-flight: xterm's callback releases each 32/64 KiB write before the next is submitted, keeping the remainder in the app queue where the 128 KiB cap can observe it instead of hiding an unbounded backlog in xterm's private WriteBuffer. While WebSocket owns terminal I/O, parallel SSE terminal/output-recovery events are discarded before JSON parsing; fallback recovery is single-flight per active session so backpressure cannot start overlapping reset+replay cycles. The route exposes capture/prepare totals in `Server-Timing`, while `[TERMINAL-PERF]` separates TTFB, body/JSON, reset+parse and total time for both selection and on-demand full pulls; parse completion is not a browser compositor/GPU paint measurement. The re-pull exists because xterm's buffer is only a WINDOW onto tmux's history and two things shrink it: tmux coalesces bursty output into pane REPAINTS that overwrite rows instead of emitting linefeeds (measured: a 60-line burst added 1 row of browser scrollback and destroyed 34), and a tab switch replays only the visible frame. tmux's own history is intact throughout — the browser just has to ask for it again. On-demand rather than automatic because at a 100k history limit the capture can be megabytes. ⚠️ **The capture ENDS with a cursor move back to the pane's own caret position** (`formatCursorRestore`, from the same `display-message` query the visible-frame path uses). The linear replay otherwise leaves the caret wherever the last character landed — the bottom-most row carrying text, which for an agent CLI is the status line — so the caret sat on the composer's border instead of its input line and every cursor-relative update the CLI sent afterwards was measured from the wrong row, until its next full redraw silently repaired it (that self-repair is why the report read as "it fixes itself as soon as Claude writes a line"). ⚠️ **The move is RELATIVE — up `rows - 1 - cursor_y`, then `\r`, then right `cursor_x` — never `CUP`.** `\x1b[<row>;<col>H` numbers rows from the top of the browser's screen, so it lands correctly only while the browser's row count equals `pane_height`, and nothing guarantees that: `resizeWindow` issues its tmux resize fire-and-forget and returns immediately, so a capture can be taken before a requested resize has applied, and `_onSessionNeedsRefresh` sends no resize at all. Counting up from the last replayed row anchors to the content both ends share. Restoring the cursor makes ROW ALIGNMENT load-bearing on this path: **no transform that can DELETE A LINE may run over a full-history capture**, because every deletion shifts the frame out from under the restored position. Four had accumulated — trailing blank rows stripped by `\n+$`, `stripInkRedrawBloat`, the `CLAUDE_BANNER_PATTERN` trim that cuts everything above the banner, and `LEADING_WHITESPACE_PATTERN` — each correct for a byte stream of successive frames and each wrong for a single rendered frame. ⚠️ **Those skips key on `isFullCapture`, meaning a capture actually came back — never on `?full=1` alone.** When `captureActivePaneBuffer` returns null (ENOBUFS, a timeout, a vanished pane, or a session with no mux at all) the reply falls back to `session.terminalBuffer`, which IS a byte stream and must still be stripped; gating on the query flag returned it whole, and a direct-PTY session takes that path on every first selection rather than only during an outage. ⚠️ A capture holding nothing visible (`hasVisibleContent`) returns `''`, because the caller reads an empty capture as "unavailable" and keeps its byte history — retaining trailing blank rows made an all-blank pane non-empty, which would have replaced real history with a blank screen from the server side, where `_replayWouldShrinkBuffer` cannot see it. ⚠️ **"One line per screen row" holds only where no row was hard-wrapped**: `-J` joins a wrapped row into its logical line (measured: a 100-character line in a 40-column pane captures as 10 lines against a 12-row pane), and the counts reconcile only once the browser xterm re-wraps at the same width — the same assumption `_estimateReplayRows` already documents. Tests: `test/tmux-capture-full-history.test.ts` covers the cursor move, the trim pairing and `hasVisibleContent`; `test/routes/session-routes.test.ts` covers a surviving blank first row, an unstripped byte-history fallback, and an empty capture leaving history intact. ⚠️ **The re-pull must never DOWNGRADE the buffer** (#205 round 2): the same reasoning that makes it a win for a shell pane makes it destructive for a repaint-mode CLI pane, where tmux keeps no history of its own (`history_size≈0` measured for a Claude pane) and the capture is roughly ONE frame while xterm may hold hundreds of rows of replayed frames — `_resetTerminalForReplay()` + rewrite then deletes history mid-scroll ("goes back a bit, repeats blocks, gets worse the further up I go"; measured A/B on a live pane: 341 rows → 42 with the guard off). `_replayWouldShrinkBuffer()` (terminal-ui.js) estimates the capture's rendered rows — escape sequences stripped, `capture-pane -J` re-wrapping accounted for — and the pull is skipped when that is more than one screen short of `buffer.active.length`. The one-screen tolerance matters: both sides are estimates (the buffer length counts trailing blank rows), so only a clear downgrade is refused. A refused session joins `_fullHistoryRepullUseless`, raising its cooldown from 4s to 60s so a hollow pane stops re-fetching megabytes on every scroll-up. Tests: `test/tmux-capture-full-history.test.ts`, `test/tmux-scrollback-eol.test.ts`, `test/terminal-scroll-routing.test.ts`, `test/terminal-flush-budget.test.ts`. +**Full-scrollback replay** (COD-164/#148, reworked for #205): `GET /api/sessions/:id/terminal?full=1` returns the ENTIRE tmux scrollback (capture-pane `-e -S -<lines>` bounded by the configured history limit, explicit `maxBuffer` from the terminal-history config, early byte-cap before normalization, CRLF-normalized for shell panes). On success the capture is returned ALONE (`source='mux-full-history'` — it supersedes the byte buffer; no duplication). The first load of each non-shell TUI session per page requests `full=1` (`_fullHistoryLoaded` Set in app.js — the old one-shot `_initialFullBufferLoad` flag was consumed by whichever tab auto-selected, leaving every other TUI tab one frame of history). Shell sessions instead load a bounded 1 MiB `?tail=` window on every selection and automatic drop recovery: a 100k-line shell capture can be tens of MiB, and automatically parsing it makes tab-switch latency scale with the entire session. Shell full history is explicit-button-only; reaching the top during an ordinary wheel/touch gesture must not reset xterm and replay the multi-megabyte capture on its main thread. Other modes may still re-pull `full=1` at the TOP, and pressing **Load full history** forces the request for any recoverably truncated session (`_maybeRefetchFullHistory`, 4s per-session gesture cooldown, in-flight + tab-switch guards, viewport position held across the replay); Shell full pulls are not retained in the tab cache, so the next switch stays bounded. Chunked replay enqueues 32 KiB pieces across safe yields, appends an xterm parse marker, then releases the live-output gate; output arriving after that release stays ordered behind the snapshot, and the marker callback supplies accurate parse timing. ⚠️ **How the load ENDS depends on where the payload came from**, and `_bufferLoadFinishOpts` (app.js) is the one place that decides it for all four fetch-and-write paths. A payload built from the server's accumulated byte history is current up to the response, so the events queued during the load already appear in it and stay DISCARDED; replaying them would duplicate output, most visibly Ink's cursor-up redraws. A pane capture (`mux-visible` or `mux-full-history`) is current only up to CAPTURE time, so `_finishBufferLoad` replays the queue from the response's own arrival timestamp (`since`) and the pre-capture events stay dropped. ⚠️ **A path that then restores a scroll position must re-take the sticky-scroll baseline** (`_syncStickyScrollBaseline`): the replay runs inside `chunkedTerminalWrite` before its promise resolves, with the terminal freshly reset, so `batchTerminalWrite` samples `_wasAtBottomBeforeWrite` as true and the next `flushPendingWrites` would scroll to the bottom over the restore. ⚠️ The cutoff is a client-side timestamp and the server broadcasts on a batch timer (8ms WebSocket, 16-50ms SSE), so a batch pending when the capture ran arrives after the response and replays although the capture holds it — bounded by one batch interval, and closable only server side by flushing that batch before the capture. Tests for the three: `test/terminal-flush-budget.test.ts` pins which sources flush, `test/terminal-buffer-flush.test.ts` pins the `since` cutoff and the baseline re-take, and `test/capture-load-window.browser.test.ts` drives both against a live server. Live output is separately one-chunk-in-flight: xterm's callback releases each 32/64 KiB write before the next is submitted, keeping the remainder in the app queue where the 128 KiB cap can observe it instead of hiding an unbounded backlog in xterm's private WriteBuffer. While WebSocket owns terminal I/O, parallel SSE terminal/output-recovery events are discarded before JSON parsing; fallback recovery is single-flight per active session so backpressure cannot start overlapping reset+replay cycles. The route exposes capture/prepare totals in `Server-Timing`, while `[TERMINAL-PERF]` separates TTFB, body/JSON, reset+parse and total time for both selection and on-demand full pulls; parse completion is not a browser compositor/GPU paint measurement. The re-pull exists because xterm's buffer is only a WINDOW onto tmux's history and two things shrink it: tmux coalesces bursty output into pane REPAINTS that overwrite rows instead of emitting linefeeds (measured: a 60-line burst added 1 row of browser scrollback and destroyed 34), and a tab switch replays only the visible frame. tmux's own history is intact throughout — the browser just has to ask for it again. On-demand rather than automatic because at a 100k history limit the capture can be megabytes. ⚠️ **The capture ENDS with a cursor move back to the pane's own caret position** (`formatCursorRestore`, from the same `display-message` query the visible-frame path uses). The linear replay otherwise leaves the caret wherever the last character landed — the bottom-most row carrying text, which for an agent CLI is the status line — so the caret sat on the composer's border instead of its input line and every cursor-relative update the CLI sent afterwards was measured from the wrong row, until its next full redraw silently repaired it (that self-repair is why the report read as "it fixes itself as soon as Claude writes a line"). ⚠️ **The move is RELATIVE — up `rows - 1 - cursor_y`, then `\r`, then right `cursor_x` — never `CUP`.** `\x1b[<row>;<col>H` numbers rows from the top of the browser's screen, so it lands correctly only while the browser's row count equals `pane_height`, and nothing guarantees that: `resizeWindow` issues its tmux resize fire-and-forget and returns immediately, so a capture can be taken before a requested resize has applied, and `_onSessionNeedsRefresh` sends no resize at all. Counting up from the last replayed row anchors to the content both ends share. Restoring the cursor makes ROW ALIGNMENT load-bearing on this path: **no transform that can DELETE A LINE may run over a full-history capture**, because every deletion shifts the frame out from under the restored position. Four had accumulated — trailing blank rows stripped by `\n+$`, `stripInkRedrawBloat`, the `CLAUDE_BANNER_PATTERN` trim that cuts everything above the banner, and `LEADING_WHITESPACE_PATTERN` — each correct for a byte stream of successive frames and each wrong for a single rendered frame. ⚠️ **Those skips key on `isFullCapture`, meaning a capture actually came back — never on `?full=1` alone.** When `captureActivePaneBuffer` returns null (ENOBUFS, a timeout, a vanished pane, or a session with no mux at all) the reply falls back to `session.terminalBuffer`, which IS a byte stream and must still be stripped; gating on the query flag returned it whole, and a direct-PTY session takes that path on every first selection rather than only during an outage. ⚠️ A capture holding nothing visible (`hasVisibleContent`) returns `''`, because the caller reads an empty capture as "unavailable" and keeps its byte history — retaining trailing blank rows made an all-blank pane non-empty, which would have replaced real history with a blank screen from the server side, where `_replayWouldShrinkBuffer` cannot see it. ⚠️ **"One line per screen row" holds only where no row was hard-wrapped**: `-J` joins a wrapped row into its logical line (measured: a 100-character line in a 40-column pane captures as 10 lines against a 12-row pane), and the counts reconcile only once the browser xterm re-wraps at the same width — the same assumption `_estimateReplayRows` already documents. Tests: `test/tmux-capture-full-history.test.ts` covers the cursor move, the trim pairing and `hasVisibleContent`; `test/routes/session-routes.test.ts` covers a surviving blank first row, an unstripped byte-history fallback, and an empty capture leaving history intact. ⚠️ **The re-pull must never DOWNGRADE the buffer** (#205 round 2): the same reasoning that makes it a win for a shell pane makes it destructive for a repaint-mode CLI pane, where tmux keeps no history of its own (`history_size≈0` measured for a Claude pane) and the capture is roughly ONE frame while xterm may hold hundreds of rows of replayed frames — `_resetTerminalForReplay()` + rewrite then deletes history mid-scroll ("goes back a bit, repeats blocks, gets worse the further up I go"; measured A/B on a live pane: 341 rows → 42 with the guard off). `_replayWouldShrinkBuffer()` (terminal-ui.js) estimates the capture's rendered rows — escape sequences stripped, `capture-pane -J` re-wrapping accounted for — and the pull is skipped when that is more than one screen short of `buffer.active.length`. The one-screen tolerance matters: both sides are estimates (the buffer length counts trailing blank rows), so only a clear downgrade is refused. A refused session joins `_fullHistoryRepullUseless`, raising its cooldown from 4s to 60s so a hollow pane stops re-fetching megabytes on every scroll-up. Tests: `test/tmux-capture-full-history.test.ts`, `test/tmux-scrollback-eol.test.ts`, `test/terminal-scroll-routing.test.ts`, `test/terminal-flush-budget.test.ts`. ### Terminal scrollback: strip flavors and wheel/touch forwarding diff --git a/src/web/public/app.js b/src/web/public/app.js index ab303da5..e96e78b7 100644 --- a/src/web/public/app.js +++ b/src/web/public/app.js @@ -1921,6 +1921,16 @@ class CodemanApp { * the moment the response arrived, compared only against other client-side * readings, so there is no clock skew to worry about. * + * What this cutoff does NOT cover: the server appends output to the byte + * buffer and emits it in the same tick, but it BROADCASTS on a batch timer — + * 8ms over WebSocket, 16 to 50ms over SSE. The terminal route runs + * synchronously from `capture-pane` to its return, so a batch that was + * already pending when the capture ran leaves the server after the reply, + * arrives after `headersReceivedAt`, and is replayed although the capture + * holds it. The duplicate is one batch interval wide, against a recovery + * window that spans the whole chunked write. Closing it belongs on the + * server: flush that session's pending batch before taking the capture. + * * @param {{source?: string}} payload - The parsed `data` of a terminal response. * @param {number} headersReceivedAt - When that response reached this client. * @returns {{flushQueued: boolean, since: number}} Options for `_finishBufferLoad`. @@ -2557,6 +2567,10 @@ class CodemanApp { }); if (target === null || typeof this.terminal.scrollToLine !== 'function') this.terminal.scrollToBottom(); else this.terminal.scrollToLine(target); + // The load's own replay sampled the sticky-scroll baseline while the + // terminal sat at the bottom of a just-rewritten buffer, so the next + // flush would scroll back down and undo the restore above. + this._syncStickyScrollBaseline(); // Re-position local echo overlay at new prompt location this._localEchoOverlay?.rerender(); // Resize PTY to match actual browser dimensions (critical for OpenCode @@ -5842,6 +5856,12 @@ class CodemanApp { const delta = parsedBufferLength - rowsBefore; if (delta > 0) this.terminal.scrollToLine(delta); else this.terminal.scrollToTop(); + // The load's own replay sampled the sticky-scroll baseline while the + // terminal sat at the bottom of a just-rewritten buffer, so the next + // flush would scroll back down and undo the restore above. This path is + // reached only from a scroll-up gesture, so being dragged down is the + // exact opposite of what the user asked for. + this._syncStickyScrollBaseline(); timing.totalMs = performance.now() - requestStartedAt; this._recordTerminalLoadTiming(timing); } catch { diff --git a/src/web/public/terminal-ui.js b/src/web/public/terminal-ui.js index 1a265d61..b56f5510 100644 --- a/src/web/public/terminal-ui.js +++ b/src/web/public/terminal-ui.js @@ -3131,6 +3131,27 @@ Object.assign(CodemanApp.prototype, { return buffer.viewportY >= buffer.baseY - 2; }, + /** + * Re-take the sticky-scroll baseline from where the viewport now sits. + * + * `batchTerminalWrite` samples `_wasAtBottomBeforeWrite` before it queues + * data, and `flushPendingWrites` scrolls to the bottom off that sample. A + * buffer load that replays its queue samples at the worst possible moment: + * `_finishBufferLoad` runs inside `chunkedTerminalWrite`, before its promise + * resolves, with the terminal freshly reset and rewritten, so the sample is + * always true. A caller that then restores the reader's position would have + * that restore undone by the next flush. + * + * Every caller that scrolls the viewport somewhere other than the bottom + * after a load must call this, so the baseline describes the position the + * caller chose. `selectSession` and `_onSessionClearTerminal` deliberately + * end at the bottom, so for them the sampled true is already the truth and + * they do not call it. + */ + _syncStickyScrollBaseline() { + this._wasAtBottomBeforeWrite = this.isTerminalAtBottom(); + }, + // Record manual scroll gestures so sticky-scroll can give an upward scroll a // short grace window (see _hasRecentUserScrollUp). A downward scroll that // lands back at the bottom clears the suppression immediately. diff --git a/test/capture-load-window.browser.test.ts b/test/capture-load-window.browser.test.ts index 9f1ee810..8cea77ef 100644 --- a/test/capture-load-window.browser.test.ts +++ b/test/capture-load-window.browser.test.ts @@ -165,6 +165,9 @@ describe('output emitted during a capture load', () => { context = await browser.newContext({ viewport: { width: 1280, height: 800 } }); page = await context.newPage(); const sessionId = await openSession(page); + // Without this, a failed create passes the zero-hit assertion below + // vacuously — nothing was loaded, so nothing was replayed. + expect(sessionId).toBeTruthy(); expect(await runLoad(page, sessionId, 'history')).toBe(0); diff --git a/test/terminal-buffer-flush.test.ts b/test/terminal-buffer-flush.test.ts index 0500a3e6..36f652ef 100644 --- a/test/terminal-buffer-flush.test.ts +++ b/test/terminal-buffer-flush.test.ts @@ -84,6 +84,50 @@ function makeApp() { return { app, writes }; } +/** + * A stub carrying the REAL `batchTerminalWrite` on top of the real begin/finish + * methods, so a replay samples the sticky-scroll baseline exactly as it does in + * the browser. The terminal is a fake whose `buffer.active` the test moves by + * hand, which is what a caller's `scrollToLine` does to a real one. + */ +function makeScrollApp() { + const buffer = { viewportY: 0, baseY: 100 }; + const app = { + buffer, + terminal: { buffer: { active: buffer } }, + sessions: new Map(), + activeSessionId: null, + pendingWrites: [] as string[], + writeFrameScheduled: false, + _wasAtBottomBeforeWrite: false, + _bufferLoadSeq: 0, + _bufferLoadOwner: null as string | null, + _isLoadingBuffer: false, + _loadBufferQueue: null as { at: number; data: string }[] | null, + _scheduleTerminalWriteFlush: vi.fn(), + batchTerminalWrite: mixin.batchTerminalWrite as (data: string) => void, + isTerminalAtBottom: mixin.isTerminalAtBottom as () => boolean, + _syncStickyScrollBaseline: mixin._syncStickyScrollBaseline as () => void, + _beginBufferLoad: mixin._beginBufferLoad as BufferLoadApp['_beginBufferLoad'], + _finishBufferLoad: mixin._finishBufferLoad as BufferLoadApp['_finishBufferLoad'], + }; + return app; +} + +/** + * Slice one class method out of app.js, from its header to the next method's. + * + * Bounding the slice matters: the two methods checked below are not followed by + * a JSDoc block, so a scan for the next comment would run on into unrelated + * code and match its scroll calls instead of theirs. + */ +function methodBody(source: string, method: string): string { + const start = source.search(new RegExp(`^ {2}(?:async )?${method}\\(`, 'm')); + expect(start, `${method} not found in app.js`).toBeGreaterThan(-1); + const next = /^ {2}(?:async )?[A-Za-z_$][\w$]*\(/m.exec(source.slice(start + 1)); + return next ? source.slice(start, start + 1 + next.index) : source.slice(start); +} + /** * Simulate a live SSE event arriving while a buffer load is in progress. * Mirrors batchTerminalWrite's queue branch, which stamps each entry with its @@ -241,6 +285,60 @@ describe('buffer-load flush (COD-144)', () => { expect(writes).toEqual(['belongs-to-this-load']); }); + // ── The sticky-scroll baseline across a replay ── + // + // `batchTerminalWrite` samples `_wasAtBottomBeforeWrite` before queueing, and + // `flushPendingWrites` scrolls to the bottom off that sample. The replay runs + // inside `chunkedTerminalWrite` before its promise resolves, with the terminal + // freshly reset and rewritten, so the sample is always true. A caller that + // then restores the reader's position would have that restore undone. + + it('the replay latches the baseline true, and the viewport restore re-takes it', () => { + const app = makeScrollApp(); + const owner = app._beginBufferLoad('load-scroll'); + pushWhileLoading(app as unknown as BufferLoadApp, 'output-after-the-capture', 100); + + // The load ends with the terminal reset and rewritten, so it reads as bottom. + app.buffer.viewportY = app.buffer.baseY; + app._finishBufferLoad(owner, { flushQueued: true, since: 0 }); + expect(app._wasAtBottomBeforeWrite).toBe(true); + + // The caller now puts the reader back where they were reading. + app.buffer.viewportY = 40; + app._syncStickyScrollBaseline(); + + // The next flush must leave them there. + expect(app._wasAtBottomBeforeWrite).toBe(false); + }); + + it('a restore that lands back at the bottom keeps sticky scroll armed', () => { + const app = makeScrollApp(); + const owner = app._beginBufferLoad('load-scroll-bottom'); + pushWhileLoading(app as unknown as BufferLoadApp, 'output-after-the-capture', 100); + + app.buffer.viewportY = app.buffer.baseY; + app._finishBufferLoad(owner, { flushQueued: true, since: 0 }); + app._syncStickyScrollBaseline(); + + // A reader who was already at the bottom still wants to be carried along. + expect(app._wasAtBottomBeforeWrite).toBe(true); + }); + + it('both callers that restore a scroll position re-take the baseline', () => { + // The wiring lives in app.js, outside this file's vm harness. Without it the + // two methods below restore the viewport and the next flush undoes it. + const source = readFileSync(resolve(import.meta.dirname, '../src/web/public/app.js'), 'utf8'); + + for (const method of ['_onSessionNeedsRefresh', '_maybeRefetchFullHistory']) { + const body = methodBody(source, method); + const restoreAt = body.lastIndexOf('scrollToLine('); + const syncAt = body.indexOf('this._syncStickyScrollBaseline()'); + expect(restoreAt, `${method} no longer restores a scroll position`).toBeGreaterThan(-1); + expect(syncAt, `${method} never re-takes the baseline`).toBeGreaterThan(-1); + expect(syncAt, `${method} re-takes the baseline before its restore`).toBeGreaterThan(restoreAt); + } + }); + it('empty queue + flushQueued is a no-op (no throw, no writes)', () => { const { app, writes } = makeApp(); const owner = app._beginBufferLoad('load-empty'); diff --git a/test/terminal-flush-budget.test.ts b/test/terminal-flush-budget.test.ts index 514e93da..aa103661 100644 --- a/test/terminal-flush-budget.test.ts +++ b/test/terminal-flush-budget.test.ts @@ -322,6 +322,51 @@ describe('terminal flush budget', () => { expect(app._bufferLoadOwner).toBe(null); }); + // ── Which payloads end their load by replaying the queue ── + // + // A pane capture is current only up to capture time, so the tail that arrived + // after the response exists nowhere else and has to be replayed. The server's + // accumulated byte history is current up to the response, so replaying on top + // of it would duplicate output. `_bufferLoadFinishOpts` is the one place that + // decides this, for all four paths that fetch a terminal buffer and write it. + + it('replays the tail for a visible-pane capture', () => { + const { CodemanApp } = loadAppHarness(); + const app = Object.create(CodemanApp.prototype) as any; + + expect(app._bufferLoadFinishOpts({ source: 'mux-visible' }, 1234)).toEqual({ + flushQueued: true, + since: 1234, + }); + }); + + it('replays the tail for a full-history capture', () => { + const { CodemanApp } = loadAppHarness(); + const app = Object.create(CodemanApp.prototype) as any; + + expect(app._bufferLoadFinishOpts({ source: 'mux-full-history' }, 1234)).toEqual({ + flushQueued: true, + since: 1234, + }); + }); + + it('discards the queue for the accumulated byte history', () => { + const { CodemanApp } = loadAppHarness(); + const app = Object.create(CodemanApp.prototype) as any; + + expect(app._bufferLoadFinishOpts({ source: 'history' }, 1234).flushQueued).toBe(false); + }); + + it('discards the queue for a payload that names no source', () => { + // Fails toward the safe answer: a duplicated Ink redraw corrupts the screen, + // while a dropped tail is repaired by the CLI's next full repaint. + const { CodemanApp } = loadAppHarness(); + const app = Object.create(CodemanApp.prototype) as any; + + expect(app._bufferLoadFinishOpts({}, 1234).flushQueued).toBe(false); + expect(app._bufferLoadFinishOpts(undefined, 1234).flushQueued).toBe(false); + }); + it('does not snap back to bottom during Codex Working redraws right after the user scrolls up', () => { const { app } = loadTerminalUiHarness('codex'); const scrollToBottom = vi.fn(); From cfd771d1d8121e80791b8cc9cf148853c5b845d1 Mon Sep 17 00:00:00 2001 From: Michael Grundberg <michael.grundberg@irisity.com> Date: Fri, 18 Sep 2026 18:44:04 +0200 Subject: [PATCH 26/28] test(terminal): pin all four buffer-load paths to the shared flush helper MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The first version of this fix decided the flush policy in `selectSession` alone, and a later pass found it still covering one path of four. Nothing in the CI gate stops a fifth path, or an inlined `{ flushQueued: true }`, from splitting that policy up again — the browser suite that would notice is excluded from `npm test`. A static scan over `selectSession`, `_onSessionNeedsRefresh`, `_onSessionClearTerminal` and `_maybeRefetchFullHistory` asserts each one asks `_bufferLoadFinishOpts`, reusing the `methodBody` slice the sticky-scroll guard already needed. Verified by inlining the policy back into `_onSessionClearTerminal`, which fails it by name. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- test/terminal-buffer-flush.test.ts | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/test/terminal-buffer-flush.test.ts b/test/terminal-buffer-flush.test.ts index 36f652ef..9bc61771 100644 --- a/test/terminal-buffer-flush.test.ts +++ b/test/terminal-buffer-flush.test.ts @@ -339,6 +339,26 @@ describe('buffer-load flush (COD-144)', () => { } }); + it('every path that fetches a terminal buffer and writes it asks the shared helper', () => { + // Drift guard. The first version of this fix covered one of the four paths, + // and a later pass found it still covering one of four. Nothing else in the + // gate stops a fifth path, or an inlined `{ flushQueued: true }`, from + // splitting the policy up again; the browser suite that would notice does + // not run in CI. + const source = readFileSync(resolve(import.meta.dirname, '../src/web/public/app.js'), 'utf8'); + + for (const method of [ + 'selectSession', + '_onSessionNeedsRefresh', + '_onSessionClearTerminal', + '_maybeRefetchFullHistory', + ]) { + expect(methodBody(source, method), `${method} decides the flush policy itself`).toContain( + 'this._bufferLoadFinishOpts(' + ); + } + }); + it('empty queue + flushQueued is a no-op (no throw, no writes)', () => { const { app, writes } = makeApp(); const owner = app._beginBufferLoad('load-empty'); From 3730bc7df5279c073b23f82f2eb86f8913abe993 Mon Sep 17 00:00:00 2001 From: Michael Grundberg <michael.grundberg@irisity.com> Date: Fri, 18 Sep 2026 18:56:15 +0200 Subject: [PATCH 27/28] docs(terminal): correct what selectSession does with the viewport MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The JSDoc on `_syncStickyScrollBaseline` said `selectSession` deliberately ends at the bottom, so the baseline the replay samples is already true there. It does not. `selectSession` calls `scrollToBottom()` after the write and then ends at `scrollToLastNonEmptyLine()` (app.js:6512), which targets `lastNonEmptyLine - rows + 2` and therefore parks ABOVE `baseY` whenever the replayed frame keeps trailing blank rows — which a full capture does on purpose, since no transform that can delete a line may run over one. Its baseline really is a stale true. What covers it is the sticky snap itself: since de864e7d that snap fires only when the flush found the viewport already at the bottom (`preserveViewportY === null`), which a parked selectSession viewport is not. That commit landed on master after this branch was cut, so the guard arrives with the merge rather than being present here. `_onSessionClearTerminal` is unchanged in the comment and was correct: it resets and rewrites with no scroll afterwards, so it does end at the bottom. Comment only; no behaviour change. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --- src/web/public/terminal-ui.js | 21 ++++++++++++++++----- 1 file changed, 16 insertions(+), 5 deletions(-) diff --git a/src/web/public/terminal-ui.js b/src/web/public/terminal-ui.js index b56f5510..9b56d0aa 100644 --- a/src/web/public/terminal-ui.js +++ b/src/web/public/terminal-ui.js @@ -3142,11 +3142,22 @@ Object.assign(CodemanApp.prototype, { * always true. A caller that then restores the reader's position would have * that restore undone by the next flush. * - * Every caller that scrolls the viewport somewhere other than the bottom - * after a load must call this, so the baseline describes the position the - * caller chose. `selectSession` and `_onSessionClearTerminal` deliberately - * end at the bottom, so for them the sampled true is already the truth and - * they do not call it. + * `_onSessionNeedsRefresh` and `_maybeRefetchFullHistory` restore a position + * and both call this, so their baseline describes the position they chose. + * + * The other two load paths do not call it, for different reasons. + * `_onSessionClearTerminal` resets and rewrites with no scroll afterwards, + * so the sampled true is already the truth there. `selectSession` does NOT + * end at the bottom, whatever its `scrollToBottom()` after the write + * suggests: it ends at `scrollToLastNonEmptyLine()`, which targets + * `lastNonEmptyLine - rows + 2` and therefore parks ABOVE `baseY` whenever + * the replayed frame keeps trailing blank rows, which a full capture does on + * purpose. Its baseline is a stale true. What decides whether that matters + * is the sticky snap in `flushPendingWrites`, and since de864e7d that snap + * fires only when the flush found the viewport already at the bottom + * (`preserveViewportY === null`), which a parked selectSession viewport is + * not. Do not read the absent call here as a claim that selectSession lands + * at the bottom. */ _syncStickyScrollBaseline() { this._wasAtBottomBeforeWrite = this.isTerminalAtBottom(); From 3cdb4bf42ecb2028b9a2c5aaa2b181bac7343630 Mon Sep 17 00:00:00 2001 From: Codeman maintainer <noreply@anthropic.com> Date: Fri, 18 Sep 2026 21:38:40 +0200 Subject: [PATCH 28/28] docs(terminal): the merge-time notes promised on #436 The four edits the review said would be folded in at merge, none of them code: the changeset becomes one user-facing paragraph, since it is what CHANGELOG.md and the release notes print; the `_bufferLoadFinishOpts` comment now names the second contributor to the duplicate window (`captureActivePaneBuffer` is `execSync`, so anything painted into the pane before the server read it is in the capture and is broadcast after the reply) and says why a `history` payload keeps the pre-existing discard when its exposure is the same; the `_finishBufferLoad` doc block moves from above `_beginBufferLoad` onto the function it documents; and the test file's header describes both rules the file now pins instead of only COD-144. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> --- ...y-output-that-arrived-after-the-capture.md | 48 +------------------ src/web/public/app.js | 37 +++++++++----- src/web/public/terminal-ui.js | 39 +++++++++------ test/terminal-buffer-flush.test.ts | 39 +++++++++------ 4 files changed, 75 insertions(+), 88 deletions(-) diff --git a/.changeset/fix-replay-output-that-arrived-after-the-capture.md b/.changeset/fix-replay-output-that-arrived-after-the-capture.md index 0cb24af0..023623ef 100644 --- a/.changeset/fix-replay-output-that-arrived-after-the-capture.md +++ b/.changeset/fix-replay-output-that-arrived-after-the-capture.md @@ -2,50 +2,4 @@ "aicodeman": patch --- -fix(terminal): keep the output a pane capture could not contain - -Live terminal events are queued while a buffer load runs, and the load discards -that queue when it ends. That is right when the loaded buffer is the server's -accumulated byte history: the route appends to that history right up to the -moment it serializes the response, so the queued events already appear in it and -replaying them would duplicate output. - -A tmux pane capture is a photograph, current only as of the instant -`capture-pane` ran. Output printed afterwards was queued and then dropped, with -nothing scheduling a re-fetch, and the CLI's next partial redraw landed on a -frame the terminal never received. A `?full=1` load returns the capture alone, -so it lost everything from the capture to the end of the chunked write. A -`?tail=` load carries the byte history in front of the capture, so it lost -everything from the response to the end of that write. A shell session shows -this most plainly, because its output is linear and nothing repaints it. - -Queue entries now carry their arrival time, and `_finishBufferLoad` takes a -`since` cutoff so a capture load replays exactly the tail that arrived after the -response headers. All four paths that fetch a terminal buffer and write it use -the same rule, through one shared `_bufferLoadFinishOpts` helper: selecting a -session, the backpressure refresh, the clear-terminal reload, and the -full-history re-pull. The backpressure refresh matters most, because it exists -to restore output the client already dropped once and could drop more while -doing it. - -Two things had to change for that tail to still exist when the load ends. -`chunkedTerminalWrite` is what ends the load for any non-empty buffer, so it -takes the flush policy and applies it at its own finish sites. -`_beginBufferLoad` no longer empties the queue when the same load re-enters it, -which it does on every write, because that reset discarded the fetch window -before anything could replay it. - -A path that replays its queue and then restores a scroll position re-takes the -sticky-scroll baseline (`_syncStickyScrollBaseline`). The replay runs with the -terminal freshly reset, so it reads as sitting at the bottom, and the next flush -would scroll there and undo the restore. The backpressure refresh and the -full-history re-pull are the two paths that restore a position, and both are -ones a reader reaches while scrolled up. - -One duplicate window stays open and is not closable from the browser. The server -appends output to the byte buffer in the same tick it emits, but broadcasts on a -batch timer, 8ms over WebSocket and 16 to 50ms over SSE. A batch already pending -when `capture-pane` ran therefore leaves the server after the reply and is -replayed although the capture holds it. It is one batch interval wide, against a -recovery window that spans the whole chunked write, and closing it means -flushing that session's pending batch before taking the capture. +fix(terminal): keep the output a pane capture could not contain. Opening a session, a backpressure refresh, a clear-terminal reload and a full-history re-pull all load the screen from a tmux pane capture, and anything the CLI printed between that capture and the end of the load used to be dropped, so its next partial redraw landed on a frame the terminal had never seen: missing or garbled output right after a tab switch or a refresh, plainest in a shell session. Each load now replays exactly the output that arrived after the capture, through one shared rule for all four paths, and a refresh that restores your scroll position no longer snaps back to the bottom afterwards. diff --git a/src/web/public/app.js b/src/web/public/app.js index 88fb115d..bb40152d 100644 --- a/src/web/public/app.js +++ b/src/web/public/app.js @@ -1917,23 +1917,36 @@ class CodemanApp { * browser after the response headers can already be in it. Such a load * replays exactly that tail; discarding it drops the CLI's output for the * rest of the load window, and its next partial redraw then lands on a frame - * the terminal never received. A payload built from the server's accumulated - * byte history needs the opposite: that history is current up to the - * response, so replaying the queue on top of it would duplicate output. + * the terminal never received. + * + * A `history` payload is the server's byte buffer alone: the direct-PTY + * fallback, or a mux pane whose capture came back empty. The route reads + * that buffer in the same synchronous tick it takes the capture, so it is + * current up to the route's own read and no further, which is the same + * exposure. It deliberately keeps the pre-existing discard all the same: + * both cases are rare, neither has been measured, and a duplicated Ink + * redraw is more visible than a few milliseconds of missing output. + * `capturedFromMux` below is the one line to widen if either turns out to + * matter. * * `headersReceivedAt` is the caller's own `performance.now()` reading from * the moment the response arrived, compared only against other client-side * readings, so there is no clock skew to worry about. * - * What this cutoff does NOT cover: the server appends output to the byte - * buffer and emits it in the same tick, but it BROADCASTS on a batch timer — - * 8ms over WebSocket, 16 to 50ms over SSE. The terminal route runs - * synchronously from `capture-pane` to its return, so a batch that was - * already pending when the capture ran leaves the server after the reply, - * arrives after `headersReceivedAt`, and is replayed although the capture - * holds it. The duplicate is one batch interval wide, against a recovery - * window that spans the whole chunked write. Closing it belongs on the - * server: flush that session's pending batch before taking the capture. + * What this cutoff does NOT cover, and there are two contributors. The + * server appends output to the byte buffer and emits it in the same tick, + * but BROADCASTS on a batch timer (8ms over WebSocket, 16 to 50ms over SSE), + * and the terminal route runs synchronously from `capture-pane` to its + * return, so a batch already pending when the capture ran leaves the server + * after the reply, arrives after `headersReceivedAt`, and is replayed + * although the capture holds it. Separately, `captureActivePaneBuffer` is + * `execSync`, which blocks the event loop for the whole capture: anything + * tmux had already painted into the pane that the server had not yet read + * from the attach PTY is in the capture too, is broadcast only after the + * reply, and replays the same way. The duplicate is one batch interval plus + * one capture wide, against a recovery window that spans the whole chunked + * write. Closing it belongs on the server: flush that session's pending + * batch before taking the capture. * * @param {{source?: string}} payload - The parsed `data` of a terminal response. * @param {number} headersReceivedAt - When that response reached this client. diff --git a/src/web/public/terminal-ui.js b/src/web/public/terminal-ui.js index 5c1af2e7..c767ad36 100644 --- a/src/web/public/terminal-ui.js +++ b/src/web/public/terminal-ui.js @@ -3924,6 +3924,30 @@ Object.assign(CodemanApp.prototype, { }); }, + /** + * Open a buffer load: live terminal events are queued from here until + * `_finishBufferLoad` decides what to do with them. Returns the load token the + * finish call must present; a stale token makes that call a no-op. + * + * @param {string} [owner] Reuse an existing token to re-enter the same load + * (see below); omit it to start a new one. + * @returns {string} The load token. + */ + _beginBufferLoad(owner) { + if (this._bufferLoadSeq === undefined) this._bufferLoadSeq = 0; + const loadOwner = owner === undefined ? `buffer-${++this._bufferLoadSeq}` : owner; + // `selectSession` opens the load before its fetch, and `chunkedTerminalWrite` + // opens it again under the SAME owner when it starts writing. Resetting the + // queue on that second call would throw away everything that arrived during + // the fetch, which on the capture path is output no buffer holds. Re-entering + // one load keeps its queue; a genuinely new load still starts empty. + const reentering = this._bufferLoadOwner === loadOwner && Array.isArray(this._loadBufferQueue); + this._bufferLoadOwner = loadOwner; + this._isLoadingBuffer = true; + if (!reentering) this._loadBufferQueue = []; + return loadOwner; + }, + /** * Complete a buffer load: unblock live SSE writes. * Called when chunkedTerminalWrite finishes (or is skipped for empty buffers). @@ -3960,21 +3984,6 @@ Object.assign(CodemanApp.prototype, { * is true, replay queued events whose arrival timestamp is at or after * `since` (default 0, meaning the whole queue). */ - _beginBufferLoad(owner) { - if (this._bufferLoadSeq === undefined) this._bufferLoadSeq = 0; - const loadOwner = owner === undefined ? `buffer-${++this._bufferLoadSeq}` : owner; - // `selectSession` opens the load before its fetch, and `chunkedTerminalWrite` - // opens it again under the SAME owner when it starts writing. Resetting the - // queue on that second call would throw away everything that arrived during - // the fetch, which on the capture path is output no buffer holds. Re-entering - // one load keeps its queue; a genuinely new load still starts empty. - const reentering = this._bufferLoadOwner === loadOwner && Array.isArray(this._loadBufferQueue); - this._bufferLoadOwner = loadOwner; - this._isLoadingBuffer = true; - if (!reentering) this._loadBufferQueue = []; - return loadOwner; - }, - _finishBufferLoad(owner, opts) { if (owner !== undefined && this._bufferLoadOwner !== owner) { return false; diff --git a/test/terminal-buffer-flush.test.ts b/test/terminal-buffer-flush.test.ts index 9bc61771..bacd2923 100644 --- a/test/terminal-buffer-flush.test.ts +++ b/test/terminal-buffer-flush.test.ts @@ -1,20 +1,31 @@ /** - * @fileoverview Regression tests for the buffer-load flush path (COD-144). + * @fileoverview Regression tests for the buffer-load flush path: what becomes of + * the live terminal events queued while a buffer load runs, once the load ends. * - * Bug: newly launched Shell sessions rendered BLANK until a tab-switch. The - * buffer-load path (`selectSession` → `_beginBufferLoad`/`_finishBufferLoad`) - * QUEUES live SSE terminal events while `_isLoadingBuffer` is true, then on - * completion DISCARDS the queue (`_loadBufferQueue = null`). That de-dup is - * correct for an established session (the fetched buffer already contains the - * queued output, so replaying it would duplicate Ink redraws). But for a - * brand-new shell the fetch resolves BEFORE the PTY emits its prompt — the - * fetched buffer is empty and the prompt arrives only as a queued event, which - * then gets discarded → blank terminal. + * Two rules, each from a real bug. * - * Fix: `_finishBufferLoad(owner, { flushQueued })` REPLAYS the queued events - * through `batchTerminalWrite()` (after `_isLoadingBuffer` is cleared, so they - * write through normally) ONLY when the load painted nothing. The default path - * (no opts) still discards, preserving de-dup for established sessions. + * COD-144: newly launched Shell sessions rendered BLANK until a tab-switch. The + * load path (`selectSession` → `_beginBufferLoad`/`_finishBufferLoad`) queues + * live events while `_isLoadingBuffer` is true and used to DISCARD the queue on + * completion. Right for a buffer built from the server's byte history (the + * queued output is already in it, so replaying it duplicates Ink redraws), + * wrong for a brand-new shell whose fetch resolves BEFORE the PTY emits its + * prompt: the prompt arrived only as a queued event and was thrown away. A + * caller that knows the load painted nothing passes `{ flushQueued: true }` + * and the queue is REPLAYED through `batchTerminalWrite()` after + * `_isLoadingBuffer` is cleared, so the events write through normally. + * + * #436: a tmux pane capture is current only as of the instant `capture-pane` + * ran, so everything the CLI printed between the capture and the end of the + * chunked write was queued and dropped, and its next partial redraw landed on + * a frame the terminal never received. Queue entries now carry their arrival + * time and `_finishBufferLoad` takes a `since` cutoff, so a capture load + * replays exactly the tail that arrived after the response headers. All four + * fetch-and-write paths take that policy from one helper, + * `_bufferLoadFinishOpts`, and a static scan below pins each of them to it, + * because the same fix had already been written into one path out of four, + * twice. A path that replays and then restores a scroll position re-takes the + * sticky-scroll baseline (`_syncStickyScrollBaseline`), pinned the same way. * * Loaded via `vm` with a stubbed context (no jsdom — jsdom is broken on this * box; see connection-indicator.test.ts). We extract the REAL