diff --git a/.changeset/fix-report-the-captured-pane-geometry.md b/.changeset/fix-report-the-captured-pane-geometry.md index c973e359..06ddcc38 100644 --- a/.changeset/fix-report-the-captured-pane-geometry.md +++ b/.changeset/fix-report-the-captured-pane-geometry.md @@ -4,44 +4,11 @@ fix(terminal): replay a pane capture at the geometry it was taken at -A visible-frame capture repaints each row at an absolute position, counting up -to the pane's height and out to the pane's width. A terminal shorter than that -clamps every address past its own height onto its last line, so the overflow -rows overwrite one another and the rows underneath are lost. Against a 50-row -pane, a 30-row terminal rendered 28 of a 45-line command and drew the surviving -frame twice. A narrower terminal damages the same frame a second way: each row -is painted out to the pane's own width, so the browser wraps every painted row, -and the wrap on the last one scrolls the whole frame up by a row. - -Nothing in the response said what geometry the frame was built for, so the -client could not detect either case. A capture now reports the geometry it was -really taken at through `capturedGeometry` on `PaneCaptureOptions`, and the -terminal response carries it as `captureCols` and `captureRows`. Both fields are -absent unless the response really carries a capture, since a body that was never -positioned has no geometry to describe. When a captured pane is taller or wider -than the terminal, or the size that produced the capture did not survive the -load, `selectSession` replays once at the size that stuck. - -That comparison runs on a visible-frame response only. A full-history response -is linear scrollback closed by a relative cursor move, and a byte-history -response carries no row alignment at all, so a size mismatch damages neither and -a replay repairs neither. The distinction matters because the first load of -every non-shell session per page takes the full-history path, where a replay -would capture the whole tmux scrollback a second time. - -Two guards keep the replay to the one pass that can converge. `resizeRetry` caps -it at a single attempt, so two competing fits cannot trade replays forever. A -pane already drawing at the size the client just requested is left alone, which -is the signature of a clamp rather than a race: `getTerminalDimensions()` floors -at 40x10 while `fitAddon.fit()` does not, so a terminal narrower than 40 columns -or shorter than 10 rows reports a pane permanently bigger than itself and would -otherwise replay on every tab switch without ever converging. - -One case is still reported rather than repaired. A pane can be too tall because -`Session.resize` declined the resize outright, which it does for a small -viewport while a desktop viewport's size claim is live. The retry re-sends the -same declined resize and captures the same pane, so it costs the one capped -attempt and the frame is shown as it is. Repairing it means deciding who owns -the pane size while a desktop claim is live, which is a policy question this -does not touch. The reported geometry still helps, because the client can see -the mismatch at all rather than being blind to it. +Opening a session could draw a frame built for a pane bigger than your terminal. A +taller pane wrote its overflow rows onto the last line and lost the rows underneath +(against a 50-row pane, a 30-row terminal rendered 28 of a 45-line command and drew +the survivors twice), and a wider one wrapped every row and scrolled the whole frame +up by one. The terminal response now reports the geometry the capture was really +taken at, so the browser can see the mismatch and replay once at the size that stuck. +A pane that cannot be sized to fit is diagnosed once per session instead of on every +tab switch. diff --git a/.changeset/run-instance-count-non-claude.md b/.changeset/run-instance-count-non-claude.md new file mode 100644 index 00000000..b0ff890f --- /dev/null +++ b/.changeset/run-instance-count-non-claude.md @@ -0,0 +1,11 @@ +--- +"aicodeman": patch +--- + +fix(run): make the Instance count stepper work for every non-Claude mode + +The Instance count stepper next to the Run button only ever applied to Claude. +Setting it to 3 and launching OpenCode, Codex, Gemini, Antigravity, Pi, OMP, Grok or +DeepSeek started exactly one session, with no error and no hint that the control had +done nothing. All eight now launch the count you asked for, and the opening banner +says how many are starting. diff --git a/CLAUDE.md b/CLAUDE.md index a3d779ea..7b525ecd 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -257,7 +257,7 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph **Circuit breakers**: the Ralph breaker prevents respawn thrashing (`CLOSED` → `HALF_OPEN` → `OPEN`; reset via `/api/sessions/:id/ralph-circuit-breaker/reset`). **Distinct: the PTY-exit breaker** (`session-pty-exit-breaker.ts`) trips after repeated rapid PTY exits and blocks auto-restarts. ⚠️ It resets ONLY via an explicit `{clearBreaker:true}` body on `POST /api/sessions/:id/interactive`; the frontend's auto-reattach in `selectSession()` sends no body and must never clear it. → [architecture-invariants#circuit-breakers-ralph--pty-exit](docs/architecture-invariants.md#circuit-breakers-ralph-and-pty-exit) -**Full-scrollback replay**: `GET /api/sessions/:id/terminal?full=1` returns the entire tmux scrollback, bounded by the configured history limit. On success the capture is returned ALONE (`source='mux-full-history'`), superseding the byte buffer so nothing duplicates. The first load of each non-shell TUI session per page requests `full=1` (`_fullHistoryLoaded` Set); Shell selection and automatic drop recovery always use a bounded 1 MiB `?tail=` window. Shell loads the rest only when **Load full history** is pressed; ordinary scrolling must not trigger a multi-megabyte reset+replay on xterm's main thread. Other modes may re-pull at the TOP (cooldown-guarded — tmux repaints bursty output in place, so browser scrollback shrinks while tmux's history stays complete). Live writes are one-chunk-in-flight, released by xterm's parse callback, so xterm's private queue cannot bypass the browser's 128 KiB render cap. While WebSocket owns terminal I/O, duplicate SSE terminal events are dropped before JSON parsing, and recovery is single-flight per active session. ⚠️ **A `full=1` capture ENDS with a cursor move back to the pane's own caret position**, counted UP from the last replayed row — without it the caret stays where the last character landed, which for an agent CLI is the status line, and every cursor-relative update the CLI sends afterwards is measured from the wrong row. The move is relative, not `CUP`: absolute row addressing is only right while the browser's rows equal the pane's, and `resizeWindow` does not wait for tmux, so a capture can be taken before a requested resize applies. That makes row alignment load-bearing on this path: no transform that can DELETE A LINE may run over the capture, so it keeps its trailing blank rows and skips redraw-bloat stripping, the banner trim and the leading-whitespace strip. ⚠️ Those three skips key on whether a capture actually CAME BACK (`isFullCapture`), never on `?full=1` alone — the fallback to the byte history is a stream of successive frames that must still be stripped, and a session with no mux takes it on every load. A capture holding nothing visible returns '' so the byte history survives instead of a blank screen replacing it. ⚠️ A full re-pull must never DOWNGRADE the buffer: a repaint-mode CLI pane keeps no tmux history, so its capture is one frame and the reset+rewrite would delete history mid-scroll — `_replayWouldShrinkBuffer()` refuses it and slows that session's cooldown to 60s. → [architecture-invariants#full-scrollback-replay](docs/architecture-invariants.md#full-scrollback-replay) +**Full-scrollback replay**: `GET /api/sessions/:id/terminal?full=1` returns the entire tmux scrollback, bounded by the configured history limit. On success the capture is returned ALONE (`source='mux-full-history'`), superseding the byte buffer so nothing duplicates. The first load of each non-shell TUI session per page requests `full=1` (`_fullHistoryLoaded` Set); Shell selection and automatic drop recovery always use a bounded 1 MiB `?tail=` window. Shell loads the rest only when **Load full history** is pressed; ordinary scrolling must not trigger a multi-megabyte reset+replay on xterm's main thread. Other modes may re-pull at the TOP (cooldown-guarded — tmux repaints bursty output in place, so browser scrollback shrinks while tmux's history stays complete). Live writes are one-chunk-in-flight, released by xterm's parse callback, so xterm's private queue cannot bypass the browser's 128 KiB render cap. While WebSocket owns terminal I/O, duplicate SSE terminal events are dropped before JSON parsing, and recovery is single-flight per active session. ⚠️ **A `full=1` capture ENDS with a cursor move back to the pane's own caret position**, counted UP from the last replayed row — without it the caret stays where the last character landed, which for an agent CLI is the status line, and every cursor-relative update the CLI sends afterwards is measured from the wrong row. The move is relative, not `CUP`: absolute row addressing is only right while the browser's rows equal the pane's, and `resizeWindow` does not wait for tmux, so a capture can be taken before a requested resize applies. That makes row alignment load-bearing on this path: no transform that can DELETE A LINE may run over the capture, so it keeps its trailing blank rows and skips redraw-bloat stripping, the banner trim and the leading-whitespace strip. ⚠️ Those three skips key on whether a capture actually CAME BACK (`isFullCapture`), never on `?full=1` alone — the fallback to the byte history is a stream of successive frames that must still be stripped, and a session with no mux takes it on every load. A capture holding nothing visible returns '' so the byte history survives instead of a blank screen replacing it. ⚠️ A full re-pull must never DOWNGRADE the buffer: a repaint-mode CLI pane keeps no tmux history, so its capture is one frame and the reset+rewrite would delete history mid-scroll — `_replayWouldShrinkBuffer()` refuses it and slows that session's cooldown to 60s. ⚠️ **A visible capture now REPORTS the geometry it was taken at** (`captureCols`/`captureRows`, #435), because a frame built for a pane taller or wider than the browser is damaged two ways at once (overflow rows clamp onto the last line; a narrower browser wraps every painted row) and nothing in the response used to say so. Both fields are ABSENT when no frame was positioned, so every consumer tests `Number.isFinite`, never truthiness: a `display-message` cursor query that fails makes `capturePaneBuffer` return the raw capture while the route still labels it `mux-visible`. The comparison runs on `mux-visible` ONLY, the replay is capped at one attempt, and a pane that cannot be sized to fit latches in `_geometryRetryUseless` so it is diagnosed once per session rather than on every tab switch. → [architecture-invariants#full-scrollback-replay](docs/architecture-invariants.md#full-scrollback-replay) **Terminal touch gestures: link taps and text selection**: on a touch device xterm's own handlers see neither — `touch-action: none` plus touchstart's preventDefault suppress the browser's compatibility mouse events, `_installMobileTapMouseGuard` drops the trusted ones that still arrive, and the synthetic `mousedown`/`mouseup` pair dispatched for mouse REPORTING goes to the `.xterm` root, an ANCESTOR of the screen element the linkifier and SelectionService listen on. So both gestures are driven explicitly. ⚠️ **A tap activates the link under it** through the SAME provider that feeds the hover linkifier (`_terminalLinkAtPoint`, containment mirroring xterm's `_linkAtPosition`), synchronously inside `touchend` — that is what keeps the user gesture `window.open` needs — and BEFORE any mouse report, mirroring `_handleDesktopTerminalClick`'s skip for a hovered link. Two rows keep their meaning: the caret's logical line (`_tapIsOnCaretLine`, where a tap places the cursor in text the USER typed) and TUI-owned rows (`_isActionableMobileTerminalTap`, answering a dialog). ⚠️ The caret line is the boundary rather than the tap INTENT, because a shell classifies every tap as `'input'` and gating on that would leave every URL in shell output inert. ⚠️ **Long-press selects** by driving xterm's public `select()` (renderer-independent — under WebGL the glyphs are pixels and native selection cannot exist), drag or a further tap extends, and Copy goes through `copyTerminalSelection()` for its execCommand fallback on plain-HTTP installs. Three guards are load-bearing and each came from a real phone: the compat mouse pair after `touchend` (xterm focuses on mousedown and SelectionService resets the model there, so the keyboard sprang up and the selection vanished on lift), the platform's own ~500ms long-press (Android Chrome focuses the nearest editable element — the helper textarea — through no event a handler can preventDefault, so a bounded focus guard blurs it and `contextmenu` is suppressed for the gesture window), and `copyTerminalSelection()`'s closing `terminal.focus()` (right on desktop, wrong on a phone). Tests: `test/terminal-touch-tap.test.ts`. diff --git a/docs/architecture-invariants.md b/docs/architecture-invariants.md index afcf92b1..a50d34b8 100644 --- a/docs/architecture-invariants.md +++ b/docs/architecture-invariants.md @@ -134,6 +134,8 @@ Tests: `test/docker-hosts.test.ts`, `test/docker-exec-options.test.ts`, `test/do **Full-scrollback replay** (COD-164/#148, reworked for #205): `GET /api/sessions/:id/terminal?full=1` returns the ENTIRE tmux scrollback (capture-pane `-e -S -` bounded by the configured history limit, explicit `maxBuffer` from the terminal-history config, early byte-cap before normalization, CRLF-normalized for shell panes). On success the capture is returned ALONE (`source='mux-full-history'` — it supersedes the byte buffer; no duplication). The first load of each non-shell TUI session per page requests `full=1` (`_fullHistoryLoaded` Set in app.js — the old one-shot `_initialFullBufferLoad` flag was consumed by whichever tab auto-selected, leaving every other TUI tab one frame of history). Shell sessions instead load a bounded 1 MiB `?tail=` window on every selection and automatic drop recovery: a 100k-line shell capture can be tens of MiB, and automatically parsing it makes tab-switch latency scale with the entire session. Shell full history is explicit-button-only; reaching the top during an ordinary wheel/touch gesture must not reset xterm and replay the multi-megabyte capture on its main thread. Other modes may still re-pull `full=1` at the TOP, and pressing **Load full history** forces the request for any recoverably truncated session (`_maybeRefetchFullHistory`, 4s per-session gesture cooldown, in-flight + tab-switch guards, viewport position held across the replay); Shell full pulls are not retained in the tab cache, so the next switch stays bounded. Chunked replay enqueues 32 KiB pieces across safe yields, appends an xterm parse marker, then releases the live-output gate; output arriving after that release stays ordered behind the snapshot, and the marker callback supplies accurate parse timing. ⚠️ **How the load ENDS depends on where the payload came from**, and `_bufferLoadFinishOpts` (app.js) is the one place that decides it for all four fetch-and-write paths. A payload built from the server's accumulated byte history is current up to the response, so the events queued during the load already appear in it and stay DISCARDED; replaying them would duplicate output, most visibly Ink's cursor-up redraws. A pane capture (`mux-visible` or `mux-full-history`) is current only up to CAPTURE time, so `_finishBufferLoad` replays the queue from the response's own arrival timestamp (`since`) and the pre-capture events stay dropped. ⚠️ **A path that then restores a scroll position must re-take the sticky-scroll baseline** (`_syncStickyScrollBaseline`): the replay runs inside `chunkedTerminalWrite` before its promise resolves, with the terminal freshly reset, so `batchTerminalWrite` samples `_wasAtBottomBeforeWrite` as true and the next `flushPendingWrites` would scroll to the bottom over the restore. ⚠️ The cutoff is a client-side timestamp and the server broadcasts on a batch timer (8ms WebSocket, 16-50ms SSE), so a batch pending when the capture ran arrives after the response and replays although the capture holds it — bounded by one batch interval, and closable only server side by flushing that batch before the capture. Tests for the three: `test/terminal-flush-budget.test.ts` pins which sources flush, `test/terminal-buffer-flush.test.ts` pins the `since` cutoff and the baseline re-take, and `test/capture-load-window.browser.test.ts` drives both against a live server. Live output is separately one-chunk-in-flight: xterm's callback releases each 32/64 KiB write before the next is submitted, keeping the remainder in the app queue where the 128 KiB cap can observe it instead of hiding an unbounded backlog in xterm's private WriteBuffer. While WebSocket owns terminal I/O, parallel SSE terminal/output-recovery events are discarded before JSON parsing; fallback recovery is single-flight per active session so backpressure cannot start overlapping reset+replay cycles. The route exposes capture/prepare totals in `Server-Timing`, while `[TERMINAL-PERF]` separates TTFB, body/JSON, reset+parse and total time for both selection and on-demand full pulls; parse completion is not a browser compositor/GPU paint measurement. The re-pull exists because xterm's buffer is only a WINDOW onto tmux's history and two things shrink it: tmux coalesces bursty output into pane REPAINTS that overwrite rows instead of emitting linefeeds (measured: a 60-line burst added 1 row of browser scrollback and destroyed 34), and a tab switch replays only the visible frame. tmux's own history is intact throughout — the browser just has to ask for it again. On-demand rather than automatic because at a 100k history limit the capture can be megabytes. ⚠️ **The capture ENDS with a cursor move back to the pane's own caret position** (`formatCursorRestore`, from the same `display-message` query the visible-frame path uses). The linear replay otherwise leaves the caret wherever the last character landed — the bottom-most row carrying text, which for an agent CLI is the status line — so the caret sat on the composer's border instead of its input line and every cursor-relative update the CLI sent afterwards was measured from the wrong row, until its next full redraw silently repaired it (that self-repair is why the report read as "it fixes itself as soon as Claude writes a line"). ⚠️ **The move is RELATIVE — up `rows - 1 - cursor_y`, then `\r`, then right `cursor_x` — never `CUP`.** `\x1b[;H` numbers rows from the top of the browser's screen, so it lands correctly only while the browser's row count equals `pane_height`, and nothing guarantees that: `resizeWindow` issues its tmux resize fire-and-forget and returns immediately, so a capture can be taken before a requested resize has applied, and `_onSessionNeedsRefresh` sends no resize at all. Counting up from the last replayed row anchors to the content both ends share. Restoring the cursor makes ROW ALIGNMENT load-bearing on this path: **no transform that can DELETE A LINE may run over a full-history capture**, because every deletion shifts the frame out from under the restored position. Four had accumulated — trailing blank rows stripped by `\n+$`, `stripInkRedrawBloat`, the `CLAUDE_BANNER_PATTERN` trim that cuts everything above the banner, and `LEADING_WHITESPACE_PATTERN` — each correct for a byte stream of successive frames and each wrong for a single rendered frame. ⚠️ **Those skips key on `isFullCapture`, meaning a capture actually came back — never on `?full=1` alone.** When `captureActivePaneBuffer` returns null (ENOBUFS, a timeout, a vanished pane, or a session with no mux at all) the reply falls back to `session.terminalBuffer`, which IS a byte stream and must still be stripped; gating on the query flag returned it whole, and a direct-PTY session takes that path on every first selection rather than only during an outage. ⚠️ A capture holding nothing visible (`hasVisibleContent`) returns `''`, because the caller reads an empty capture as "unavailable" and keeps its byte history — retaining trailing blank rows made an all-blank pane non-empty, which would have replaced real history with a blank screen from the server side, where `_replayWouldShrinkBuffer` cannot see it. ⚠️ **"One line per screen row" holds only where no row was hard-wrapped**: `-J` joins a wrapped row into its logical line (measured: a 100-character line in a 40-column pane captures as 10 lines against a 12-row pane), and the counts reconcile only once the browser xterm re-wraps at the same width — the same assumption `_estimateReplayRows` already documents. Tests: `test/tmux-capture-full-history.test.ts` covers the cursor move, the trim pairing and `hasVisibleContent`; `test/routes/session-routes.test.ts` covers a surviving blank first row, an unstripped byte-history fallback, and an empty capture leaving history intact. ⚠️ **The re-pull must never DOWNGRADE the buffer** (#205 round 2): the same reasoning that makes it a win for a shell pane makes it destructive for a repaint-mode CLI pane, where tmux keeps no history of its own (`history_size≈0` measured for a Claude pane) and the capture is roughly ONE frame while xterm may hold hundreds of rows of replayed frames — `_resetTerminalForReplay()` + rewrite then deletes history mid-scroll ("goes back a bit, repeats blocks, gets worse the further up I go"; measured A/B on a live pane: 341 rows → 42 with the guard off). `_replayWouldShrinkBuffer()` (terminal-ui.js) estimates the capture's rendered rows — escape sequences stripped, `capture-pane -J` re-wrapping accounted for — and the pull is skipped when that is more than one screen short of `buffer.active.length`. The one-screen tolerance matters: both sides are estimates (the buffer length counts trailing blank rows), so only a clear downgrade is refused. A refused session joins `_fullHistoryRepullUseless`, raising its cooldown from 4s to 60s so a hollow pane stops re-fetching megabytes on every scroll-up. Tests: `test/tmux-capture-full-history.test.ts`, `test/tmux-scrollback-eol.test.ts`, `test/terminal-scroll-routing.test.ts`, `test/terminal-flush-budget.test.ts`. +**A capture reports the geometry it was taken at** (#435): a visible frame repaints each row at an absolute position, counting up to the pane's height and out to the pane's width, so a terminal smaller than that pane damages it two ways at once. Too short and every address past the browser's own height clamps onto the last line, overwriting the rows underneath (measured: against a 50-row pane, a 30-row terminal rendered 28 of a 45-line command and drew the survivors twice). Too narrow and each row is painted out to the pane's width, so the browser wraps every painted row and the wrap on the last one scrolls the whole frame up by one. Nothing in the response used to say what geometry the frame was built for, so the client could not see either case. `PaneCaptureOptions.capturedGeometry` carries it out, and the terminal response publishes it as `captureCols`/`captureRows`. ⚠️ **Both fields are ABSENT unless a frame was really positioned**, and every consumer must test `Number.isFinite` rather than truthiness: `mux-visible` is necessary but not sufficient, because when the `display-message` cursor query fails `capturePaneBuffer` skips the snapshot repaint and returns the raw capture, and the route still labels that non-empty body `mux-visible`. A body that positioned nothing has no geometry to describe and nothing to repair, so a comparison that fires there buys a second capture, a reset plus chunked rewrite, a dropped and reopened WebSocket and a discarded xterm snapshot for no gain. ⚠️ **The comparison runs on `mux-visible` ONLY.** A full-history body is linear scrollback closed by a RELATIVE cursor move, which is relative precisely so the browser's row count need not match the pane's, and a byte-history body carries no row alignment at all, so a size mismatch damages neither and a replay repairs neither. That gate matters because the first select of every non-shell session per page takes the full-history path, where an ungated comparison would fire most often on the one response it cannot help, at the price of a second whole-scrollback capture. ⚠️ **The replay is capped at one attempt and latches per session when it cannot converge.** `resizeRetry` stops two competing fits trading replays forever; a pane already drawing at the size just requested is left alone, which is the signature of a clamp rather than a race (`getTerminalDimensions()` floors at 40x10 while `fitAddon.fit()` does not, so a terminal under 40 columns or 10 rows reports a pane permanently bigger than itself and would replay on every tab switch); and a pass that still does not converge joins `_geometryRetryUseless`, so the case `Session.resize` declines outright (a small viewport while a desktop viewport's size claim is live, where the retry re-sends the same declined resize and captures the same pane) costs one attempt per session per page load instead of one per select. ⚠️ **A retry pass must not re-arm `_fullHistoryLoaded`**: it did not consume the full-history pull, and re-arming it would spend a whole-scrollback capture on the next select. That branch is currently unreachable by construction, since reaching it needs `source === 'mux-visible'` while a `full=1` pass is answered `mux-full-history` or `history`; a static test over the source is the habit this repo uses for an invariant nothing can execute. Tests: `test/capture-geometry-retry.browser.test.ts` (eight cases, five of which fail against the merge base), `test/tmux-capture-full-history.test.ts`, `test/routes/session-routes.test.ts`. + ### Terminal scrollback: strip flavors and wheel/touch forwarding **Two strip flavors, one carry** (#205, `session.ts:_handleTerminalOutput`): the FULL strip (`isAltScreenStripMode` = codex/claude/gemini) removes alt-screen toggles, `3J`, and mouse-tracking DECSETs. Every other mode (shell/opencode/antigravity/pi) gets the NARROW strip (`isMuxAltScreenOnlyStripMode`) — alt-screen toggles ONLY — and only when tmux-backed (`useMux`). Rationale: the tmux CLIENT emits `smcup` as its first bytes at attach, before any program runs, parking xterm in the scrollback-less alternate buffer for the whole session (touch scrolling no-ops; xterm's own wheel handler converts the wheel to Up/Down arrows = readline history cycling — both #205 symptoms). tmux never forwards a pane program's alt-screen toggles to its client (it repaints instead; measured — vim/less inside a pane emit zero to the client), so the only thing the narrow strip ever removes is tmux's own smcup. It keeps `3J` (a user's `clear` is a deliberate scrollback wipe) and the mouse DECSETs (tmux passes those through even with `mouse off`; stripping them would break htop/vim mouse support). ⚠️ The `useMux` gate is load-bearing: `startShell()`/`startInteractive()` fall back to a DIRECT PTY when mux creation fails, and there the inner program's own `?1049h` really does reach xterm — stripping it would break vim/less/htop for real. The replay path (`session-routes.ts`, via `session.usesMux`) applies the same narrow branch; the frontend mirror (`_shouldReportMouseToCli()`) stays claude/codex/gemini because only the FULL strip touches mouse DECSETs. The chunk-boundary carry (`_altScreenSeqCarry`) runs for both flavors. Tests: `test/claude-scrollback-strip.test.ts`. diff --git a/src/web/public/app.js b/src/web/public/app.js index bbd70205..219a5f35 100644 --- a/src/web/public/app.js +++ b/src/web/public/app.js @@ -6577,8 +6577,17 @@ class CodemanApp { // of every non-shell session per page takes the full-history path, an // ungated comparison fires most often on the one response it cannot help. const framePositionsRowsAbsolutely = data.source === 'mux-visible'; + // `mux-visible` is necessary but not sufficient: when the `display-message` + // cursor query fails, `capturePaneBuffer` skips the snapshot repaint and + // returns the raw capture, and the route still labels a non-empty body + // `mux-visible`. That body positions nothing and reports no geometry, so a + // size that moved during such a load has nothing to repair, and replaying + // would buy a second capture, a reset plus chunked rewrite, a dropped + // WebSocket and a discarded xterm snapshot for it. The two comparisons + // below already stand down on an absent field; this one has to as well. const sizeMovedUnderLoad = framePositionsRowsAbsolutely && + Number.isFinite(data.captureRows) && !!dimsAtCapture && !!dimsAfterLoad && (dimsAfterLoad.cols !== dimsAtCapture.cols || dimsAfterLoad.rows !== dimsAtCapture.rows); diff --git a/src/web/public/session-ui.js b/src/web/public/session-ui.js index 057f62e0..f6440cdc 100644 --- a/src/web/public/session-ui.js +++ b/src/web/public/session-ui.js @@ -930,7 +930,7 @@ Object.assign(CodemanApp.prototype, { async runClaude() { const caseName = document.getElementById('quickStartCase').value || 'testcase'; - const tabCount = Math.min(20, Math.max(1, parseInt(document.getElementById('tabCount').value) || 1)); + const tabCount = this._readTabCount(); const ownsLaunchTerminal = this._beginSessionLaunchStatus( `Starting ${tabCount} Claude session(s) in ${caseName}...` @@ -1257,9 +1257,14 @@ Object.assign(CodemanApp.prototype, { } }, - /** Reads the "Instance count" stepper, clamped like runClaude()'s own copy. */ + /** + * Reads the "Instance count" stepper, clamped to 1..20. Single source for every + * run*(), runClaude() included. Optional-chained because callers read it BEFORE + * their try block, to put the count in the opening banner: `#tabCount` ships + * unconditionally today, but a throw here would escape the launch-error path. + */ _readTabCount() { - return Math.min(20, Math.max(1, parseInt(document.getElementById('tabCount').value) || 1)); + return Math.min(20, Math.max(1, parseInt(document.getElementById('tabCount')?.value) || 1)); }, /** @@ -1274,9 +1279,6 @@ Object.assign(CodemanApp.prototype, { async _launchQuickStartInstances(caseName, tabCount, label, buildBody, ownsLaunchTerminal) { const startNumber = this._nextCaseSessionStartNumber(caseName); let firstSessionId = null; - if (tabCount > 1) { - this._appendSessionLaunchStatus(ownsLaunchTerminal, `Starting ${tabCount} ${label} session(s) in ${caseName}...`); - } for (let i = 0; i < tabCount; i++) { const sessionName = `w${startNumber + i}-${caseName}`; const res = await fetch('/api/quick-start', { @@ -1302,7 +1304,10 @@ Object.assign(CodemanApp.prototype, { const _runLoc = (this.cases || []).find(c => c.name === caseName)?.location; const isRemote = _runLoc === 'remote' || _runLoc === 'docker'; - const ownsLaunchTerminal = this._beginSessionLaunchStatus(`Starting OpenCode session in ${caseName}...`); + const tabCount = this._readTabCount(); + const ownsLaunchTerminal = this._beginSessionLaunchStatus( + `Starting ${tabCount} OpenCode session(s) in ${caseName}...` + ); // Focus in sync gesture context (see runClaude comment) this.terminal.focus(); @@ -1323,7 +1328,6 @@ Object.assign(CodemanApp.prototype, { // Quick-start with opencode mode (auto-allow tools by default). // No `effort` field — it's Claude-specific (OpenCode has no /effort). const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), this.loadAppSettingsFromStorage()); - const tabCount = this._readTabCount(); const firstSessionId = await this._launchQuickStartInstances( caseName, tabCount, @@ -1359,7 +1363,10 @@ Object.assign(CodemanApp.prototype, { const _runLoc = (this.cases || []).find(c => c.name === caseName)?.location; const isRemote = _runLoc === 'remote' || _runLoc === 'docker'; - const ownsLaunchTerminal = this._beginSessionLaunchStatus(`Starting Codex session in ${caseName}...`); + const tabCount = this._readTabCount(); + const ownsLaunchTerminal = this._beginSessionLaunchStatus( + `Starting ${tabCount} Codex session(s) in ${caseName}...` + ); this.terminal.focus(); try { @@ -1377,7 +1384,6 @@ Object.assign(CodemanApp.prototype, { const globalSettings = this.loadAppSettingsFromStorage(); const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), globalSettings); - const tabCount = this._readTabCount(); const firstSessionId = await this._launchQuickStartInstances( caseName, tabCount, @@ -1417,7 +1423,10 @@ Object.assign(CodemanApp.prototype, { const _runLoc = (this.cases || []).find(c => c.name === caseName)?.location; const isRemote = _runLoc === 'remote' || _runLoc === 'docker'; - const ownsLaunchTerminal = this._beginSessionLaunchStatus(`Starting Gemini session in ${caseName}...`); + const tabCount = this._readTabCount(); + const ownsLaunchTerminal = this._beginSessionLaunchStatus( + `Starting ${tabCount} Gemini session(s) in ${caseName}...` + ); this.terminal.focus(); try { @@ -1434,7 +1443,6 @@ Object.assign(CodemanApp.prototype, { } const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), this.loadAppSettingsFromStorage()); - const tabCount = this._readTabCount(); const firstSessionId = await this._launchQuickStartInstances( caseName, tabCount, @@ -1468,7 +1476,10 @@ Object.assign(CodemanApp.prototype, { const _runLoc = (this.cases || []).find(c => c.name === caseName)?.location; const isRemote = _runLoc === 'remote' || _runLoc === 'docker'; - const ownsLaunchTerminal = this._beginSessionLaunchStatus(`Starting Antigravity session in ${caseName}...`); + const tabCount = this._readTabCount(); + const ownsLaunchTerminal = this._beginSessionLaunchStatus( + `Starting ${tabCount} Antigravity session(s) in ${caseName}...` + ); this.terminal.focus(); try { @@ -1485,7 +1496,6 @@ Object.assign(CodemanApp.prototype, { } const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), this.loadAppSettingsFromStorage()); - const tabCount = this._readTabCount(); const firstSessionId = await this._launchQuickStartInstances( caseName, tabCount, @@ -1528,7 +1538,10 @@ Object.assign(CodemanApp.prototype, { const _runLoc = (this.cases || []).find(c => c.name === caseName)?.location; const isRemote = _runLoc === 'remote' || _runLoc === 'docker'; - const ownsLaunchTerminal = this._beginSessionLaunchStatus(`Starting Pi session in ${caseName}...`); + const tabCount = this._readTabCount(); + const ownsLaunchTerminal = this._beginSessionLaunchStatus( + `Starting ${tabCount} Pi session(s) in ${caseName}...` + ); this.terminal.focus(); try { @@ -1545,7 +1558,6 @@ Object.assign(CodemanApp.prototype, { } const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), this.loadAppSettingsFromStorage()); - const tabCount = this._readTabCount(); const firstSessionId = await this._launchQuickStartInstances( caseName, tabCount, @@ -1576,7 +1588,10 @@ Object.assign(CodemanApp.prototype, { const _runLoc = (this.cases || []).find(c => c.name === caseName)?.location; const isRemote = _runLoc === 'remote' || _runLoc === 'docker'; - const ownsLaunchTerminal = this._beginSessionLaunchStatus(`Starting OMP session in ${caseName}...`); + const tabCount = this._readTabCount(); + const ownsLaunchTerminal = this._beginSessionLaunchStatus( + `Starting ${tabCount} OMP session(s) in ${caseName}...` + ); this.terminal.focus(); try { @@ -1593,7 +1608,6 @@ Object.assign(CodemanApp.prototype, { } const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), this.loadAppSettingsFromStorage()); - const tabCount = this._readTabCount(); const firstSessionId = await this._launchQuickStartInstances( caseName, tabCount, @@ -1635,7 +1649,10 @@ Object.assign(CodemanApp.prototype, { const _runLoc = (this.cases || []).find(c => c.name === caseName)?.location; const isRemote = _runLoc === 'remote' || _runLoc === 'docker'; - const ownsLaunchTerminal = this._beginSessionLaunchStatus(`Starting Grok session in ${caseName}...`); + const tabCount = this._readTabCount(); + const ownsLaunchTerminal = this._beginSessionLaunchStatus( + `Starting ${tabCount} Grok session(s) in ${caseName}...` + ); this.terminal.focus(); try { @@ -1652,7 +1669,6 @@ Object.assign(CodemanApp.prototype, { } const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), this.loadAppSettingsFromStorage()); - const tabCount = this._readTabCount(); const firstSessionId = await this._launchQuickStartInstances( caseName, tabCount, @@ -1704,7 +1720,10 @@ Object.assign(CodemanApp.prototype, { const _runLoc = (this.cases || []).find(c => c.name === caseName)?.location; const isRemote = _runLoc === 'remote' || _runLoc === 'docker'; - const ownsLaunchTerminal = this._beginSessionLaunchStatus(`Starting DeepSeek session in ${caseName}...`); + const tabCount = this._readTabCount(); + const ownsLaunchTerminal = this._beginSessionLaunchStatus( + `Starting ${tabCount} DeepSeek session(s) in ${caseName}...` + ); this.terminal.focus(); try { @@ -1730,7 +1749,6 @@ Object.assign(CodemanApp.prototype, { } const envOverrides = this.buildEnvOverrides(this.getCaseSettings(caseName), this.loadAppSettingsFromStorage()); - const tabCount = this._readTabCount(); const firstSessionId = await this._launchQuickStartInstances( caseName, tabCount, diff --git a/test/run-mode-ui.test.ts b/test/run-mode-ui.test.ts index c90113d4..79d6080e 100644 --- a/test/run-mode-ui.test.ts +++ b/test/run-mode-ui.test.ts @@ -1122,4 +1122,67 @@ describe('Grok quick start', () => { expect(requests).toEqual(['/api/grok/status']); expect(errors[0]).toContain('https://x.ai/cli/install.sh'); }); + + // The whole point of the shared _launchQuickStartInstances() helper. Before it, + // every non-Claude run*() hardcoded exactly one quick-start call, so the + // "Instance count" stepper next to the Run button silently did nothing on all + // eight of them: no error, no hint, just the wrong number of sessions. The five + // fixture edits that came with the change stub tabCount at '1', so they pass + // identically with and without it; this is the one that does not. + it('launches tabCount sessions with sequential w- names and selects the first', async () => { + const elements: Record = { + quickStartCase: { value: 'grok-case' }, + tabCount: { value: '3' }, + }; + const requests: Array<{ url: string; body?: any }> = []; + const CodemanApp = function CodemanApp(this: any) {}; + let created = 0; + + const context = vm.createContext({ + CodemanApp, + localStorage: { getItem: () => null, setItem: () => {} }, + document: { getElementById: (id: string) => elements[id] ?? null }, + fetch: async (url: string, init?: { body?: string }) => { + const body = init?.body ? JSON.parse(init.body) : undefined; + requests.push({ url, body }); + if (url === '/api/grok/status') + return { + json: async () => ({ + success: true, + data: { available: true, path: '/home/user/.grok/bin', version: '1.0.5' }, + }), + }; + if (url === '/api/quick-start') { + const id = `sess-gk-${created++}`; + return { + json: async () => ({ success: true, data: { sessionId: id, session: { id, name: body.sessionName } } }), + }; + } + throw new Error(`unexpected fetch: ${url}`); + }, + console, + }); + + const sessionUi = readFileSync(resolve(import.meta.dirname, '../src/web/public/session-ui.js'), 'utf8'); + vm.runInContext(sessionUi, context, { filename: 'session-ui.js' }); + + const app = new (CodemanApp as any)(); + app.terminal = { clear: () => {}, writeln: () => {}, focus: () => {} }; + app.loadAppSettingsFromStorage = () => ({}); + app.getCaseSettings = () => ({}); + app.buildEnvOverrides = () => ({}); + app.sessions = new Map(); + app._onSessionCreated = (session: any) => app.sessions.set(session.id, session); + app._renderSessionTabsImmediate = () => {}; + const selected: string[] = []; + app.selectSession = async (id: string) => { + selected.push(id); + }; + + await app.runGrok(); + + const names = requests.filter((r) => r.url === '/api/quick-start').map((r) => r.body.sessionName); + expect(names).toEqual(['w1-grok-case', 'w2-grok-case', 'w3-grok-case']); + expect(selected).toEqual(['sess-gk-0']); + }); });