From d26f26fe34bb436a3a94ba71cbac90a82a42eab3 Mon Sep 17 00:00:00 2001 From: Codeman maintainer Date: Sun, 9 Aug 2026 01:18:03 +0200 Subject: [PATCH] chore: version packages Co-Authored-By: Claude Opus 5 (1M context) --- CHANGELOG.md | 23 + CLAUDE.md | 6 +- README.md | 78 +- docs/agent-control-plan.md | 662 ++++++++++ docs/api-reference.md | 341 ++++++ docs/architecture-invariants.md | 14 + docs/extending-codeman.md | 169 ++- package-lock.json | 4 +- package.json | 3 +- skills/codeman/SKILL.md | 274 +++++ skills/codeman/reference/endpoints.md | 214 ++++ skills/codeman/reference/recipes.md | 249 ++++ src/config/agent-wait.ts | 112 ++ src/hooks-config.ts | 18 +- src/session-cli-builder.ts | 8 +- src/tmux-manager.ts | 6 +- src/web/routes/hook-event-routes.ts | 22 + src/web/routes/session-routes.ts | 546 ++++++++- src/web/schemas.ts | 61 + src/web/server.ts | 16 + src/web/session-listener-wiring.ts | 28 + src/web/session-wait-registry.ts | 1089 +++++++++++++++++ test/hook-secret-selfheal.test.ts | 19 + test/hooks-config.test.ts | 11 + test/http-contract.test.ts | 93 ++ test/mocks/mock-session.ts | 24 +- test/routes/session-input-wait.test.ts | 671 ++++++++++ .../routes/session-wait-output-routes.test.ts | 432 +++++++ test/routes/session-wait-routes.test.ts | 811 ++++++++++++ test/session-cli-builder.test.ts | 35 + test/session-wait-registry.test.ts | 1055 ++++++++++++++++ test/tmux-manager.test.ts | 35 +- 32 files changed, 7075 insertions(+), 54 deletions(-) create mode 100644 docs/agent-control-plan.md create mode 100644 skills/codeman/SKILL.md create mode 100644 skills/codeman/reference/endpoints.md create mode 100644 skills/codeman/reference/recipes.md create mode 100644 src/config/agent-wait.ts create mode 100644 src/web/session-wait-registry.ts create mode 100644 test/routes/session-input-wait.test.ts create mode 100644 test/routes/session-wait-output-routes.test.ts create mode 100644 test/routes/session-wait-routes.test.ts create mode 100644 test/session-wait-registry.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index 262e83db..52f3eae9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,28 @@ # aicodeman +## 1.13.0 + +### Minor Changes + +- Agent wait primitives, the Codeman agent skill, a fix for hooks dying silently on HTTPS installs, and the tab-strip UX improvements from the previous batch. + + **Agent wait primitives (new API surface, the reason this is a minor).** Three bounded long-polls let an agent driving Codeman from a shell block instead of poll: + - `GET /api/v1/sessions/:id/wait` blocks until a lifecycle signal fires (`until=stop,idle,working,blocked,exit`, `fresh=1` to require a new transition). + - `GET /api/v1/sessions/:id/wait-output` blocks until a literal substring appears in the session's output (`match=`, `nocase=`, `from=now|buffer`; never regex, by design). + - `wait`/`waitTimeout` on `POST /api/v1/sessions/:id/input` (send-and-wait) registers the waiter before typing, closing the race where a separate wait reports the previous turn's idle state as this turn's answer. + + Shared semantics: a timeout is HTTP 200 with `wait.timedOut: true` (callers loop over short waits; tunnels cut idle connections), timeouts are clamped to [1s, 600s] and echoed back as `wait.timeoutMs`, all three nest the result under `data.wait`, and `status`/`limitPaused` ride along. `stop`/`blocked` exist for `claude` mode only: requesting them explicitly elsewhere is a 400, the default set silently narrows and echoes what it waited on. Capacity caps (16 waiters per session, 128 process-wide) answer 409/429, waiter slots release on client hang-up, and shutdown resolves parked waiters instead of stranding them. Bounds are operator-tunable via `CODEMAN_WAIT_*` env vars. + + Reliability details that came out of three verification rounds: a worker that dies inside its tmux pane is now detected at the mux layer (pane-death probe, ~750ms cache, a 3s watcher for waits already parked), so a corpse answers `exit` instead of `idle` and send-and-wait rolls back its dedup seq when the write went nowhere; output matching normalizes charset-designation escapes (a stock bash prompt's `ESC ( B` no longer breaks `match=tnode:`) and holds back partial escapes at chunk boundaries, so matches straddling PTY chunks are found. + + **Codeman agent skill (`skills/codeman`).** A packaged skill that teaches an agent running inside a Codeman session to drive the API safely: guard preamble (refuses outside `CODEMAN_MUX=1`, resolves credentials from the data dir `.env` or the install's service definition), self-protection (`is_self` prefix check in both directions), readiness for claude workers (composer-first, trust dialog as bounded fallback), send-and-wait loops that cannot report a never-submitted prompt as success, marker-synchronized shell flows, fan-out patterns, and cleanup discipline. Ships in the npm package via the `files` entry. + + **Hooks were dying silently on every HTTPS install (bug fix).** The generated hook curls lacked `-k`, so on `--https` installs (self-signed cert) every hook event (`stop`, `permission_prompt`, `elicitation_dialog`, `idle_prompt`, `teammate_idle`, `task_completed`) failed TLS verification and the failure was swallowed, taking respawn's definitive idle signals with it. Hooks are now generated with `curl -sk`, and a staleness detector regenerates the on-disk hook config of already-created cases the next time a session starts in them. Relatedly, `CODEMAN_API_URL` is no longer exported with a guessed `http://localhost:3000` fallback (wrong scheme on HTTPS installs); it is omitted unless the server has stamped the real URL, so in-session guards fail closed. + + **Tab strip (from the previous batch, reported by christianhaberl):** action icons (kill/pop-out) now appear on the active tab only, middle-click closes a tab, tab hover uses a fixed width with a sliding title instead of resizing the strip, and the pop-out button is opt-in (default off). + + **Docs.** `docs/api-reference.md` gained the full long-polling contract (signals by mode, readiness, what the matcher sees, response discriminators); `docs/extending-codeman.md` and the README carry verified copy-paste orchestration recipes; `docs/architecture-invariants.md` records the load-bearing ordering, liveness, and edge-triggered-signal invariants. Net +163 tests (4300 passing in the CI sweep). + ## 1.12.2 ### Patch Changes diff --git a/CLAUDE.md b/CLAUDE.md index d58ce832..4b0bf816 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -74,7 +74,7 @@ When user says "COM": CI runs `npm run check:lockfile` on every push/PR, so lockfile drift fails the build even if the `version-packages` script is bypassed. -**Version**: 1.12.2 (must match `package.json`) +**Version**: 1.13.0 (must match `package.json`) ## Project Overview @@ -180,6 +180,8 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph **Input**: `session.writeViaMux()` for programmatic/curl input via tmux `send-keys -l` + `send-keys Enter`, single-line only. Interactive **browser** input goes through a durable **exactly-once** layer: a stable `clientId` + monotonic per-session `seq` persisted to localStorage until the server ACKs, so a dropped link cannot lose or double-deliver a prompt. `ws-connection-registry.ts` supersedes only same-TAB reconnects, so two tabs on one session coexist. → [architecture-invariants#input-delivery-and-ws-resilience](docs/architecture-invariants.md#input-delivery-and-ws-resilience) +**Agent wait primitives**: bounded long-polls so an agent driving Codeman from a shell can block instead of poll: `GET /api/sessions/:id/wait` (lifecycle signal), `GET /api/sessions/:id/wait-output` (literal substring, **never** regex) and `wait`/`waitTimeout` on `POST /api/sessions/:id/input`. Registry in `session-wait-registry.ts` (pure, no `Session` reference), bounds in `config/agent-wait.ts`. ⚠️ **A timeout is a 200** (`wait.timedOut`), never an error, so callers loop over short waits. ⚠️ `stop`/`blocked` come from Claude Code hooks and therefore fire for **`claude` mode ONLY** (`shell` installs none either); asking for one explicitly on another mode is a 400, the default set silently drops them. ⚠️ Send-and-wait registers the waiter BEFORE the write (a separate POST-then-wait races and reports the PREVIOUS turn), and both teardown paths must `notifySignal('exit')` BEFORE `cancelAll()`. ⚠️ Client-hangup abort listens on **`reply.raw`** guarded by `writableFinished`: on `req.raw`, `close` fires when the request BODY ends, which on a POST killed every send-and-wait instantly and no `app.inject()` test could see it. ⚠️ Worker liveness cannot come from `session.pid` — for a tmux session that is the local attach client, which outlives a worker dying inside its pane — so it is probed at the mux layer (`isPaneDead`, ~750 ms cache) on blocking waits only, never on the input hot path. ⚠️ Signals are edge-triggered with no history: one that fires with no waiter registered is unobservable afterwards, so gather fan-outs with send-and-wait or latched `wait-output` markers, never fire-and-forget-then-sequential-signal-waits. → [architecture-invariants#agent-wait-primitives](docs/architecture-invariants.md#agent-wait-primitives), `docs/api-reference.md` + **Idle detection**: Multi-layer (completion message → AI check → output silence → token stability). See `docs/respawn-state-machine.md`. **Auto-resume on usage limit** (opt-in per session, top of the Respawn tab): when Claude halts on a subscription limit, `usage-limit-patterns.ts` (pure, unit-tested) parses the reset time and `SessionAutoOps` arms a timer for reset+2min, then sends Esc + `continue`. ⚠️ Respawn cycles are blocked while paused (`isLimitPaused` guard in `onIdleDetected`), which is what prevents `/clear` from wiping the paused conversation. Claude-mode only. → [architecture-invariants#auto-resume-on-usage-limit](docs/architecture-invariants.md#auto-resume-on-usage-limit) @@ -292,7 +294,7 @@ Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. L ### API Routes -~199 handlers across 21 route files in `src/web/routes/`: system (45), sessions (32), cases (27), files (16), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), me (2), teams (2), search (1), hooks (1), clipboard (1), status-telemetry (1), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details. +~200 handlers across 21 route files in `src/web/routes/`: system (45), sessions (34), cases (27), files (16), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), me (2), teams (2), search (1), hooks (1), clipboard (1), status-telemetry (1), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details. **HTTP contract** (stable since 0.9.x, see `docs/versioning-policy.md`; full envelope/status/error-code/SSE spec in `docs/api-reference.md`): responses use the `ApiResponse` envelope — `{ success: true, data? }` or `{ success: false, error, errorCode }` (`src/types/api.ts`). `/api/v1/*` is a versioned alias of `/api/*` (URL rewrite in `server.ts`). diff --git a/README.md b/README.md index 6c60e7c2..b6c03b94 100644 --- a/README.md +++ b/README.md @@ -680,15 +680,21 @@ When a CLI runs in a Codeman-managed session, these environment variables are se ### Rules of the road (read before you POST) -1. **Single-line input only.** Programmatic input is sent as literal text **+ Enter** in one shot. Multi-line strings break the agent TUI (Ink) — send one line, or split into multiple calls. +1. **Single-line input, ending in `\r`.** Programmatic input is sent as literal text, and Enter fires **only when the input contains a carriage return**: `{"input":"run tests\r"}`. Without the `\r` the text sits on the session's prompt unsubmitted (and a combined `wait` runs its full timeout on a turn that never started). Embedded newlines are stripped rather than rejected, so `"echo A\necho B\r"` runs the joined command `echo Aecho B`: send one line per call. 2. **Make input idempotent.** Include a stable `clientId` and a monotonic per-session `seq` on `POST …/input`. The server de-duplicates, so a retry after a dropped connection can't double-deliver a prompt. -3. **Auth.** If `CODEMAN_PASSWORD` is set, send HTTP Basic auth (user `admin` or `CODEMAN_USERNAME`) or a `codeman_session` cookie. The default loopback install is passwordless. A missing `Origin` header is allowed, so plain `curl` works; cross-site browser origins are rejected (CSRF guard). +3. **Auth.** If `CODEMAN_PASSWORD` is set, send HTTP Basic auth (user `admin` or `CODEMAN_USERNAME`) or a `codeman_session` cookie. The default loopback install is passwordless. A missing `Origin` header is allowed, so plain `curl` works; cross-site browser origins are rejected (CSRF guard). ⚠️ A `401` replies with the bare string `Unauthorized`, **not** the JSON envelope, so piping it into `jq` throws a parse error instead of showing the failure: check the status before parsing. 4. **Response envelope.** Most endpoints return `{ "success": true, "data": … }` (errors: `{ "success": false, "error", "errorCode" }`). A few legacy GETs return bare bodies — **handle both** (`body.data ?? body`). 5. **`/api/v1/*`** is a stable alias of `/api/*`. +6. **Wait instead of polling, and don't treat a timeout as an error.** The wait endpoints answer with HTTP `200` and `wait.timedOut: true` when nothing happened in time, so loop over short waits (60s is the default) rather than issuing one long call, because tunnels cut idle connections. `wait.timeoutMs` tells you the timeout the server actually applied after clamping (600s ceiling). +7. **Only `claude` sessions emit `stop` and `blocked`.** Those two come from Claude Code hooks; `shell` and the external CLIs (opencode/codex/gemini/antigravity) accept only `idle`, `working` and `exit`. Asking for `stop` explicitly on those is a `400`; omitting `until` is always safe. ⚠️ On a `shell` session `idle` fires **once**, at startup, and never again, so send-and-wait there can only time out; synchronize hook-less sessions with a `wait-output` marker. +8. **Nothing reports "ready", so wait for it explicitly.** A new session answers `{"signal":"exit","immediate":true}` (that means *not started*, not *crashed*) until its PID exists, and a `claude` worker in a fresh case then sits on the CLI's trust dialog. Prompt it there and the wait resolves on `idle` in ~2s looking exactly like a finished turn, while the text sits stuck in the dialog. Recipe 2b below is the sequence that avoids it. ### Recipes ```bash +# CODEMAN_API_URL is auto-set inside every Codeman session, correct scheme included. +# The fallback below fits a stock install; on a --https install set the https:// URL +# yourself and add -k to each curl (self-signed cert). API="${CODEMAN_API_URL:-http://127.0.0.1:3000}" # (add -u admin:"$CODEMAN_PASSWORD" to each call if a password is set) @@ -700,18 +706,63 @@ curl -s -X POST "$API/api/quick-start" \ -H 'Content-Type: application/json' \ -d '{"caseName":"refactor-auth","mode":"claude","effort":"high"}' | jq +# 2b. Wait until that worker is actually READY (see rule 8): composer marker first, +# first-run trust dialog only as the fallback. (Probing trust first and sending +# a blind Enter misfires on re-runs: the dialog text stays in the buffer forever, +# so the probe matches stale text and the Enter lands in a ready composer.) +# Match single tokens: TUI text can reach the matcher without its spaces. +until [ "$(curl -s "$API/api/sessions/$SID" | jq '.data.pid')" != null ]; do sleep 1; done +R=$(curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=bypass' \ + --data-urlencode 'from=buffer' --data-urlencode 'timeout=5000') +if ! jq -e '.data.wait.matched' <<<"$R" >/dev/null; then + T=$(curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=trust' \ + --data-urlencode 'from=buffer' --data-urlencode 'timeout=2000') + jq -e '.data.wait.matched' <<<"$T" >/dev/null && \ + curl -s -X POST "$API/api/sessions/$SID/input" -H 'Content-Type: application/json' \ + -d '{"input":"\r","useMux":true}' # accept the first-run trust dialog + curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=bypass' \ + --data-urlencode 'from=buffer' --data-urlencode 'timeout=45000' >/dev/null +fi + # 3. Send a prompt into a session (exactly-once: clientId + seq) curl -s -X POST "$API/api/sessions/$SID/input" \ -H 'Content-Type: application/json' \ - -d '{"input":"Run the test suite and summarize failures","useMux":true,"clientId":"agent-1","seq":1}' + -d '{"input":"Run the test suite and summarize failures\r","useMux":true,"clientId":"agent-1","seq":1}' -# 4. Read the terminal back -curl -s "$API/api/sessions/$SID/output" | jq -r '.data // .' +# 4. Send a prompt and BLOCK until that turn is done (registers the wait before +# writing, so it can't answer with the previous turn's idle state) +curl -s -X POST "$API/api/sessions/$SID/input" \ + -H 'Content-Type: application/json' \ + -d '{"input":"Run the test suite and summarize failures\r","useMux":true, + "clientId":"agent-1","seq":2,"wait":"stop,exit","waitTimeout":60000}' \ + | jq '.data.wait' # -> {"signal":"stop","timedOut":false,"waitedMs":41230,...} +# (`stop` is the definitive end-of-turn hook. Adding `idle` makes it resolve on a +# spinner pause too, and on anything that redraws a ❯ prompt — like a dialog.) -# 5. Stream live events (session output, agent activity, status) +# 4b. Timed out? That's a 200, not a failure. Loop over short waits. +curl -s "$API/api/sessions/$SID/wait?until=stop,exit&timeout=60000" | jq '.data.wait' + +# 4c. Or wait for a marker in the output (works for shell sessions too). +# ⚠️ Unique per call (tmux repaints replay old screen text), and SPLIT so the +# typed line never contains it: your own keystrokes echo into the output +# stream, so an unsplit marker matches before the command has run. from=buffer +# catches a marker that printed before the wait landed. +N=$RANDOM +curl -s -X POST "$API/api/sessions/$SID/input" -H 'Content-Type: application/json' \ + -d "{\"input\":\"M=DONE; npm test; echo \${M}_$N rc=\$?\r\",\"useMux\":true}" +curl -sG "$API/api/sessions/$SID/wait-output" \ + --data-urlencode "match=DONE_$N" --data-urlencode 'from=buffer' \ + --data-urlencode 'timeout=60000' | jq '.data.wait' + +# 5. Read the terminal back. ⚠️ Use terminal?tail=, NOT /output: the latter's +# textOutput is empty for every tmux-backed (i.e. every interactive) session. +# tail counts BYTES, and what comes back is terminal data, ANSI included. +curl -s "$API/api/sessions/$SID/terminal?tail=8000" | jq -r '.data.terminalBuffer' + +# 6. Stream live events (session output, agent activity, status) curl -sN "$API/api/events" # Server-Sent Events -# 6. Schedule recurring work (cron-style job) +# 7. Schedule recurring work (cron-style job) curl -s -X POST "$API/api/cron/jobs" \ -H 'Content-Type: application/json' \ -d '{"name":"nightly-deps","agentType":"claude","workingDir":"/home/me/proj", @@ -719,11 +770,11 @@ curl -s -X POST "$API/api/cron/jobs" \ "inputMode":"typed","scheduleType":"daily","dailyTime":"03:00", "enabled":true,"concurrencyPolicy":"warn_only"}' | jq -# 7. Inspect background sub-agents and their transcripts +# 8. Inspect background sub-agents and their transcripts curl -s "$API/api/subagents" | jq '.data // .' curl -s "$API/api/subagents/$AID/transcript" | jq -r '.data // .' -# 8. Whole-system snapshot (sessions, settings, respawn, stats) +# 9. Whole-system snapshot (sessions, settings, respawn, stats) curl -s "$API/api/status" | jq ``` @@ -749,7 +800,7 @@ Codeman registers Claude Code hooks that `POST /api/hook-event` (`permission_pro ## API -REST over Fastify — **~190 handlers across 20 route modules**, plus an SSE stream and a WebSocket terminal channel. All responses use the `ApiResponse` envelope (`{success, data}` / `{success, error, errorCode}`); `/api/v1/*` is a stable alias. A representative subset: +REST over Fastify — **~200 handlers across 21 route modules**, plus an SSE stream and a WebSocket terminal channel. All responses use the `ApiResponse` envelope (`{success, data}` / `{success, error, errorCode}`); `/api/v1/*` is a stable alias. A representative subset: ### Sessions @@ -757,8 +808,11 @@ REST over Fastify — **~190 handlers across 20 route modules**, plus an SSE str | -------- | -------------------------- | ---------------------------------------------------------------------------------- | | `GET` | `/api/sessions` | List all | | `POST` | `/api/quick-start` | Create case + start session (`{caseName?, mode?, effort?, envOverrides?}`) | -| `POST` | `/api/sessions/:id/input` | Send input (`{input, useMux?, clientId?, seq?}` — `clientId`+`seq` = exactly-once) | -| `GET` | `/api/sessions/:id/output` | Read terminal output | +| `POST` | `/api/sessions/:id/input` | Send input (`{input, useMux?, clientId?, seq?, wait?, waitTimeout?}`: `clientId`+`seq` = exactly-once; `wait` blocks until the turn ends) | +| `GET` | `/api/sessions/:id/terminal` | Read terminal output (`?tail=`, `?full=1`); the read path for interactive sessions | +| `GET` | `/api/sessions/:id/output` | Parsed one-shot output (`textOutput` is empty for tmux-backed sessions) | +| `GET` | `/api/sessions/:id/wait` | Block until a signal fires (`?until=stop,idle,exit&timeout=&fresh=`); a timeout is a `200` | +| `GET` | `/api/sessions/:id/wait-output` | Block until a literal string appears (`?match=&nocase=&from=now\|buffer&timeout=`) | | `GET` | `/api/sessions/unified` | Unified live + history list (Session Manager) — `?q=&limit=` | | `POST` | `/api/sessions/:id/pin` | Pin/unpin in the Session Manager (`{pinned}`) | | `PUT` | `/api/session-order` | Sync tab order across devices (`{order: [ids]}`) | diff --git a/docs/agent-control-plan.md b/docs/agent-control-plan.md new file mode 100644 index 00000000..1e1ae474 --- /dev/null +++ b/docs/agent-control-plan.md @@ -0,0 +1,662 @@ +# Agent Control Plan: skill packaging + wait primitives + +**Status**: steps 1 to 5 IMPLEMENTED and multi-round verified, uncommitted as of 2026-08-08. +Step 6 (CLI install command + per-case injection + `agentSkillEnabled`) is not built. +See [§7 Build log](#7-build-log-what-actually-happened) for what shipped, what each +verification round found, and what is still open. + +**Date**: 2026-08-08 +**Scope**: Part 1 (agent skill) and Part 2 (wait primitives) were specified and built. +Parts 3 to 5 are captured so they are not lost, but remain deliberately deferred. + +--- + +## 0. Where this came from: what herdr does + +[herdr](https://github.com/herdrdev/herdr) (Rust, Apache-2.0, ~25.8k stars) is a terminal +multiplexer built around AI coding agents. Relevant findings from the research pass: + +| Capability | How herdr does it | +| --------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Agent state | Four states (`idle`, `working`, `blocked`, `done`) that roll up pane to tab to workspace in a sidebar | +| Detection | Lifecycle hooks where the agent supports them (it names Pi and MastraCode), otherwise TOML manifests matched against a live bottom-buffer snapshot. Bundled manifests plus remote updates from herdr.dev, local overrides win | +| Control API | Newline-delimited JSON over a Unix socket (`~/.config/herdr/sessions//herdr.sock`), `{"id":"req_1","method":"pane.split","params":{}}`, dot-notation methods, plus long-lived event subscriptions | +| Discoverability | `herdr api schema` prints a machine-readable schema | +| Agent skill | `npx skills add herdrdev/herdr --skill herdr -g`, a SKILL.md wrapping the CLI, guarded by `test "${HERDR_ENV:-}" = 1` so an agent outside a herdr pane refuses to act | +| Persistence | Background server, detach with `ctrl+b q`, snapshot restore of workspaces/tabs/panes/cwd/layout, experimental screen-history replay, agent resume via native session ids, live PTY handoff across server replacement | +| Plugins | `herdr-plugin.toml` manifest, actions, event hooks, plugin panes, link handlers, GitHub-topic marketplace index | + +The commands the skill teaches the agent: + +| Group | Commands | +| --------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| workspace | `workspace list`, `workspace create` | +| tab | `tab list --workspace `, `tab create` | +| pane | `pane current`, `pane list`, `pane layout`, `pane split --current --direction right --cwd --no-focus`, `pane run ""`, `pane wait-output --match/--regex

--timeout `, `pane read --source visible\|recent\|detection` | +| agent | `agent list`, `agent start --kind --pane `, `agent prompt "" --wait --timeout `, `agent wait --until --timeout `, `agent send-keys`, `agent get`, `agent read` | + +### The honest comparison + +herdr and Codeman are not the same product. herdr is a local, keyboard-first multiplexer with +no server, no web UI, and no autonomy layer. Codeman is a server with a browser and mobile UI, +remote and Docker cases, respawn, Ralph, cron, and the orchestrator, none of which herdr has. + +What herdr genuinely does better is being **callable by the agent running inside it**. For +Codeman that is a packaging problem plus one missing primitive, not an architecture problem. + +--- + +## 1. Gap analysis + +| herdr capability | Codeman equivalent today | Gap | +| ---------------------------- | ------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------- | +| `pane split` + `agent start` | `POST /api/quick-start`, `POST /api/sessions` | none, already there | +| `agent prompt` | `POST /api/sessions/:id/input` with `clientId`+`seq` exactly-once | no `--wait` | +| `pane read` | `GET /api/sessions/:id/output`, `GET /api/sessions/:id/terminal?full=1` | none | +| `agent list` / `agent get` | `GET /api/sessions`, `GET /api/sessions/unified`, `GET /api/status` | none | +| `agent wait --until ` | SSE only (`/api/events`) | **missing**, and SSE is impractical from a shell tool | +| `pane wait-output --match` | nothing | **missing** | +| Skill file | README section "Driving Codeman from an Agent" | **not packaged**, an agent will never find it | +| Env guard `HERDR_ENV=1` | `CODEMAN_MUX=1`, `CODEMAN_API_URL`, `CODEMAN_SESSION_ID` already exported at spawn | none, the guard variables exist | +| `blocked` state | hook events (`permission_prompt`, `elicitation_dialog`) plus CSS classes plus the phone overview NEEDS YOU section | not in the wire contract (`SessionStatus = 'idle' \| 'busy' \| 'stopped' \| 'error'`) | +| `api schema` | hand-written `docs/api-reference.md` | no machine-readable schema | +| Detection manifests | hardcoded in `usage-limit-patterns.ts`, `respawn-*-patterns`, `regex-patterns.ts` | patterns are code, not data | +| Plugin runtime | deliberately refused, see `docs/extending-codeman.md` | not a gap, a decision | +| Session handoff on restart | tmux owns the PTYs, so they already survive a Codeman restart | not a gap, solved by architecture | + +**Conclusion**: roughly 90% of the capability surface already exists. Parts 1 and 2 below close +the two real gaps. + +--- + +## 2. Part 1: the Codeman agent skill + +### 2.1 Goal + +An agent running inside a Codeman session can discover and correctly drive Codeman without the +user pasting API docs into the prompt, and without inventing dangerous calls. + +### 2.2 Layout and distribution + +The `npx skills` CLI (vercel-labs/skills) clones a GitHub repo and looks for +`skills//SKILL.md`. Claude Code natively discovers `.claude/skills//SKILL.md` in a +project and `~/.claude/skills/` globally. Both are satisfied with one source of truth plus a +symlink, which is the pattern this repo already uses for `remotion-best-practices`. + +``` +skills/ + codeman/ + SKILL.md <- single source of truth + reference/ + endpoints.md <- full endpoint tables, loaded on demand + recipes.md <- worked multi-session orchestration examples +.claude/skills/codeman -> ../../skills/codeman (symlink, dogfooding in this repo) +``` + +Adding a `skills/` directory to the repo root costs one entry in the GitHub listing. CLAUDE.md +keeps the root short on purpose, so this needs a conscious sign-off; the alternative is +`docs/skills/codeman/` with a `--skill` path argument, which breaks the one-liner install. +**Recommendation**: accept `skills/` at the root, because the install one-liner is the whole +point of shipping a skill. + +Install paths, in order of how a user gets it: + +1. `npx skills add Ark0N/Codeman --skill codeman -g` (global, any agent, matches the herdr flow). +2. `codeman skill install [--global | --case ]`, a new CLI subcommand writing the same + file. This is the path for users who installed via npm and never cloned the repo. +3. **Automatic per-case injection**, modeled exactly on `applyStatusLineConfig(casePath, enabled)` + in `hooks-config.ts`: write `/.claude/skills/codeman/SKILL.md` at case creation, + gated on a new setting. Codeman already writes `/.claude/settings.local.json` hooks + through `writeHooksConfig()`, so this is the same mechanism with the same lifecycle. + +Setting name: `agentSkillEnabled`. Synced (not per-device), since it changes on-disk case +content rather than display. Default: **ON after the dogfooding phase, OFF in the first +release**. Rationale for starting OFF: Claude Code loads every skill's name and description +into context on every turn, so an always-on skill has a small permanent token cost, and we +should measure that we are buying something with it first. + +### 2.3 SKILL.md content + +Frontmatter, per the skills convention (`name` + `description` required): + +```yaml +--- +name: codeman +description: >- + Control Codeman, the session manager this agent is running inside: list sessions, + start worker sessions, send prompts, read terminal output, and wait for other agents + to finish. Only usable when CODEMAN_MUX=1. +--- +``` + +Body sections, in order: + +**1. Guard (first thing, non-negotiable).** + +```bash +test "${CODEMAN_MUX:-}" = 1 || { echo "not inside a Codeman session"; exit 1; } +API="${CODEMAN_API_URL:?CODEMAN_API_URL not set, refusing to guess}" +SELF="${CODEMAN_SESSION_ID:-}" +``` + +If `CODEMAN_MUX` is not `1`, the agent must stop and say it is not running inside a +Codeman-managed session. Same shape as herdr's `HERDR_ENV` guard, and the variables are +already exported by `tmux-manager.buildEnvExports()`. No fallback URL when +`CODEMAN_API_URL` is unset: any guess is the wrong scheme on an HTTPS install (prod is +HTTPS with a self-signed cert, hence `curl -sk` throughout), and a server the agent +cannot identify is not one it should be driving. + +**2. Rules of the road.** Lifted and tightened from README lines 666 to 745: + +- Single-line input only. Multi-line breaks the agent TUI (Ink). +- Always send `clientId` + a monotonic `seq` on `POST .../input` so a retry cannot double-deliver. +- Envelope is `{success, data}`; a few legacy GETs are bare, so read `body.data ?? body`. +- Add `-u admin:"$CODEMAN_PASSWORD"` when a password is set. Prod is HTTPS, so `curl -sk`. +- Prefer `/api/v1/*`, the stable alias. + +**3. Safety rules (the section that does not exist anywhere today).** + +- Never act on `$CODEMAN_SESSION_ID`. That is you. +- Only `DELETE` sessions **you created in this conversation**, by exact id. Keep the list. +- Never bulk-delete, never loop a `DELETE` over `/api/sessions`. There is no undo. +- Never `tmux kill-session`, `pkill tmux`, `pkill claude`. Use the API. +- Creating a session consumes a slot against the 50-session cap. Clean up what you start. + +**4. Recipes**, each one a single copy-pasteable curl: + +| Task | Call | +| -------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| list sessions | `GET /api/v1/sessions` | +| find yourself | match ids by PREFIX of `$CODEMAN_SESSION_ID` (Docker cases truncate it to 8 chars, so an equality check never fires there) | +| start a worker | `POST /api/v1/quick-start {caseName, mode, effort}` | +| send a prompt | `POST /api/v1/sessions/:id/input {input:"…\r", useMux:true, clientId, seq}` (the trailing `\r` is what sends Enter; without it the text sits on the prompt unsubmitted) | +| send prompt and wait | `POST /api/v1/sessions/:id/input {input:"…\r", wait:"stop", waitTimeout:600000}` (Part 2) | +| wait for a worker | `GET /api/v1/sessions/:id/wait?until=stop,blocked&timeout=300000` (Part 2) | +| wait for a marker | `GET /api/v1/sessions/:id/wait-output?match=DONE_&timeout=120000` (Part 2; unique per call, per §3.3's repaint rule) | +| read output | `GET /api/v1/sessions/:id/output` | +| read full scrollback | `GET /api/v1/sessions/:id/terminal?full=1` | +| watch sub-agents | `GET /api/v1/subagents` | +| schedule work | `POST /api/v1/cron/jobs` | +| clean up | `DELETE /api/v1/sessions/:id` | + +**5. Pointer to `reference/endpoints.md`** for anything not in the table, so the always-loaded +part of the skill stays small. + +### 2.4 An ergonomics guard worth adding server-side + +The skill will tell the agent not to act on itself, but a confused agent can still try. Propose: +the skill sends `X-Codeman-Caller-Session: $CODEMAN_SESSION_ID` on every request, and the server +refuses destructive operations (`DELETE /api/sessions/:id`, kill, respawn stop) when that header +equals the target id, with a clear error. + +This is a **footgun guard, not a security control**: any caller can omit the header. Document it +as such so nobody mistakes it for a boundary. It costs about 10 lines in `route-helpers.ts`. + +### 2.5 Verification + +Per the always-end-to-end-test rule, "the skill exists" is not done. Done is: + +1. Symlink it into `.claude/skills/`, start a real throwaway Codeman session, and ask that agent + to "start a worker session that runs the test suite and tell me when it finishes". +2. Confirm from the outside that exactly one new session appeared, got the prompt, and that the + lead agent waited rather than polling in a busy loop. +3. Confirm the guard: run the same prompt in a shell with `CODEMAN_MUX` unset and confirm refusal. +4. Confirm cleanup: the worker session is deleted by exact id and no other session was touched. + +Never run this against `w1`/`w2`/`w3`. + +### 2.6 Files touched + +- `skills/codeman/SKILL.md` (new), `skills/codeman/reference/*.md` (new) +- `.claude/skills/codeman` symlink (new) +- `src/cli.ts` (new `skill install` subcommand) +- `src/hooks-config.ts` (new `applyAgentSkill(casePath, enabled)`, mirroring `applyStatusLineConfig`) +- `src/web/schemas.ts` (`agentSkillEnabled` in `SettingsUpdateSchema`, which is `.strict()`) +- `src/web/routes/system-routes.ts` (settings PUT must resolve the flag from `merged`, never + from the raw body, per the partial-PUT invariant) +- `src/web/public/settings-ui.js` + `index.html` (checkbox) +- `package.json` `files` array, so `skills/` ships to npm +- README pointer, `docs/extending-codeman.md` seam 3 pointer + +--- + +## 3. Part 2: wait primitives + +### 3.1 Goal + +Make Codeman orchestratable from a shell tool. Today the only "tell me when" channel is SSE, +which a curl-driven agent cannot practically consume: it would have to hold a streaming +connection and parse events inline. herdr solves this with blocking CLI calls. Codeman should +solve it with bounded long-poll endpoints. + +All three additions are **additive**, so the versioning policy stays intact (new endpoints and +new optional fields are non-breaking). + +### 3.2 The signal model + +A waiter resolves on the first of a set of signals. Sources that already exist: + +| Signal | Source today | +| --------- | --------------------------------------------------------------------------------------------------------------------------------------------- | +| `idle` | `Session` emits `idle` (session.ts ~1775 for Claude, ~2101 for shell), wired at `session-listener-wiring.ts:402` | +| `working` | `Session` emits `working` (session.ts ~1788), wired at `session-listener-wiring.ts:401` | +| `stop` | `POST /api/hook-event` with `event: 'stop'`, the definitive "Claude finished responding" signal already used by `controller.signalStopHook()` | +| `blocked` | `POST /api/hook-event` with `permission_prompt` or `elicitation_dialog` | +| `exit` | `Session` emits `exit` | + +`stop` is the highest-quality signal for "the turn is over" and should be the documented default +for orchestration. `idle` is heuristic: output stabilization plus prompt detection, and it can +flap mid-turn when a spinner pauses. External CLI modes (`isExternalCliMode()`) have no stop +hook at all, so for opencode/codex/gemini/antigravity only `idle`, `working` and `exit` are +available. **The skill and the docs must say which signals exist per mode**, otherwise an agent +waits forever on `stop` in a codex session. + +### 3.3 Endpoint specs + +#### A. `GET /api/sessions/:id/wait` + +| Param | Type | Default | Notes | +| --------- | ---------------------------------------------- | ---------------- | ------------------------------------------------------------ | +| `until` | comma list of `idle,working,stop,blocked,exit` | `stop,idle,exit` | resolves on first match | +| `timeout` | ms | 60000 | clamped to `MAX_WAIT_MS` (600000) | +| `fresh` | `0`/`1` | `0` | `1` requires a _transition_, ignoring the state at call time | + +Response (always 200 unless the session is missing or a cap is hit): + +```json +{ + "success": true, + "data": { + "signal": "stop", + "timedOut": false, + "immediate": false, + "ended": false, + "waitedMs": 8421, + "status": "idle", + "sessionId": "...", + "until": ["stop", "idle", "exit"], + "limitPaused": false + } +} +``` + +`until` is echoed back because the server may narrow it: `stop`/`blocked` are dropped +from the DEFAULT set for external CLI modes (asking for them EXPLICITLY is a 400 +instead, since omitting `until` must never 400). `limitPaused` tells a caller that a +timeout was expected rather than a stall worth retrying hard. + +**A timeout is not an error.** `{"timedOut": true, "signal": null}` with HTTP 200, so a caller +can loop without treating every poll boundary as a failure. Errors are reserved for +`NOT_FOUND` (unknown or not-owned session) and `SESSION_BUSY` (waiter cap exceeded). + +`immediate: true` means the session was already in the requested state and `fresh` was not set. + +#### B. `GET /api/sessions/:id/wait-output` + +| Param | Type | Default | Notes | +| --------- | ------------------------------ | -------- | --------------------------------------------------------- | +| `match` | literal string, 1 to 200 chars | required | substring match against ANSI-stripped output | +| `nocase` | `0`/`1` | `0` | case-insensitive compare | +| `from` | `now` \| `buffer` | `now` | `buffer` scans the existing text buffer first, then waits | +| `timeout` | ms | 60000 | clamped to `MAX_WAIT_MS` | + +Response: `{ matched: true, timedOut: false, snippet: "...", waitedMs }`. + +**No regex in v1, deliberately.** `search-service.ts` already avoids regex specifically so there +is no ReDoS surface, and this endpoint would be even more exposed since the pattern is attacker +supplied and the input is a live stream. herdr can offer `--regex` because Rust's regex crate is +linear-time with no backtracking; JS `RegExp` is not. If regex is wanted later, the honest +options are a length-capped subset compiled once with a match budget, or `re2`. Note it and move on. + +Implementation detail that will bite if missed: a match can straddle two PTY chunks. Keep a +carry buffer of `match.length - 1` bytes from the previous chunk and test `carry + chunk`. + +⚠️ **`from=now` does not mean "printed after you asked".** tmux repaints the visible +screen on attach, resize, or any TUI redraw, and a repaint arrives as ordinary `terminal` +data. Observed live: a marker echoed a minute earlier matched instantly on a fresh +`from=now` wait. This is inherent to a terminal multiplexer, not fixable in the registry, +so the contract is: **use a marker unique per call** (`echo DONE_$RANDOM`), never a +generic one like `BUILD OK`. The skill's recipes must show that. + +The returned snippet is whitespace-collapsed (blank runs to a single newline) for +readability only; matching runs on the raw stripped text. Without it, a real pane's +`\r\n` padding between the prompt and the match fills the whole context window with +nothing, which was the first thing the live test showed. + +#### C. `wait` on the existing input endpoint + +`POST /api/sessions/:id/input` gains two optional fields: + +```json +{ "input": "run the tests\r", "useMux": true, "clientId": "agent-1", "seq": 7, "wait": "stop", "waitTimeout": 600000 } +``` + +(The trailing `\r` is required on every input body: `sendInput` sends Enter only +when the input contains a carriage return.) + +Response gains `"wait": { "signal": "stop", "timedOut": false, "waitedMs": 41230 }`. + +This is the important one, because it closes a race the standalone `GET .../wait` cannot: between +"input delivered" and "session flips to working" there is a window where a naive +send-then-wait sees the _pre-existing_ idle state and returns instantly. The combined endpoint +**registers the waiter before writing**, so that window does not exist. This is exactly why herdr +ships `agent prompt --wait` as its own thing. + +`wait` accepts `true` (the default signal set) or the same comma grammar as `until`. +Both new fields are `.nullish()`, not `.optional()`: a third-party caller building the +body with `JSON.stringify` keeps an explicit `null` on the wire, and `.optional()` +rejects that with `INVALID_INPUT`. That gotcha has shipped as a real bug twice. + +Two behaviors to preserve carefully: + +- **`useMux` is fire-and-forget today.** The handler responds without awaiting `writeViaMux`, on + purpose (a tmux child process must not block the HTTP response). With `wait` present the + handler already has to stay open, so it can await delivery, and a `writeViaMux` failure becomes + observable for the first time. The non-wait path must keep its current fire-and-forget shape + byte for byte. +- **Duplicate suppression.** A tagged redelivery (`clientId`+`seq` already applied) returns 200 + without writing. With `wait` set it still waits, since the caller's intent is "tell me when + this settles". But it waits with `requireTransition: false`, unlike a fresh delivery: the + original turn may be long over, and requiring a new transition would block a redelivery until + timeout for no reason. Fresh delivery requires a transition, a duplicate answers from the + current state. +- **Capacity rollback.** `shouldApplyInput()` MUTATES (it records the seq), and it runs before + the waiter is registered. If registration then fails on a full pool, the handler must call + `forgetInputSeq` before returning `SESSION_BUSY`, or the caller's retry is rejected as a + duplicate and the input is lost by the very mechanism reliable delivery exists for. + +### 3.4 Module design + +New file `src/web/session-wait-registry.ts`, with the IO-free core unit-testable in isolation +(same split as `self-update.ts`): + +```ts +type WaitSignal = 'idle' | 'working' | 'stop' | 'blocked' | 'exit'; + +waitForSignal(sessionId, { until: Set, timeoutMs, requireTransition }): Promise +notifySignal(sessionId, signal: WaitSignal): void +waitForOutput(sessionId, { match, nocase, timeoutMs }): Promise +notifyOutput(sessionId, chunk: string): void +cancelAll(sessionId, reason): void +``` + +Wiring points, all existing: + +- `src/web/session-listener-wiring.ts` around lines 190 and 200 already handles `working` and + `idle` and broadcasts them. Add a `notifySignal()` call next to each broadcast, plus `exit`. +- `src/web/routes/hook-event-routes.ts` already switches on `event` for the respawn controller. + Add `notifySignal(sessionId, 'stop' | 'blocked')` in the same switch. +- Output: `notifyOutput()` rides the ALREADY-attached `terminal` listener in + session-listener-wiring.ts. An earlier draft had the registry hand out attach/detach + callbacks so a listener could be added lazily; that was deleted once it was clear no + second listener is needed at all. The cost is one Map lookup per PTY chunk, which is why + the no-waiter check comes before the ANSI strip. +- Session deletion calls `notifySignal('exit')` then `cancelAll()`, so no promise is left + hanging. Both are required: `_doCleanupSession` detaches the session's listeners BEFORE + `session.stop()`, so on a delete the PTY exit event never reaches the registry, and an + `until=exit` caller would otherwise get a bare `ended` instead of its signal. Found by + live-testing the delete path, not by the unit tests. + +Memory-leak discipline, per the 24-hour-session rules: every waiter owns a timer that is cleared +on resolve, the per-session waiter set is deleted when it empties, and the output listener is +removed with it. `test/memory-leak-prevention.test.ts` should grow a case for this. + +Caps in a new `src/config/agent-wait.ts` (limits live in `src/config/`, env-overridable): + +| Constant | Default | Why | +| ------------------------- | ------- | --------------------------------------- | +| `MAX_WAIT_MS` | 600000 | an unbounded long-poll is a socket leak | +| `DEFAULT_WAIT_MS` | 60000 | short enough to survive most proxies | +| `MAX_WAITERS_PER_SESSION` | 16 | | +| `MAX_WAITERS_TOTAL` | 128 | same reasoning as `MAX_SSE_CLIENTS` | + +Exceeding a cap returns `SESSION_BUSY`, not a silent queue. + +### 3.5 Transport concerns + +Fastify is constructed with defaults in `server.ts:329-331`. `requestTimeout` defaults to 0 +(disabled) and `keepAliveTimeout` (72s) applies between requests, not to an in-flight one, so a +10-minute in-process hold is fine. **Verify this on the real instance before relying on it.** + +Intermediaries are the actual risk. Prod is reached through `tailscale serve`, and users also run +cloudflared tunnels; both can cut an idle connection. That is why `DEFAULT_WAIT_MS` is 60s and +why the documented pattern is a client-side loop over short waits rather than one 10-minute call. +The skill's recipes must show the loop. + +### 3.6 Edge cases to get right + +| Case | Behavior | +| ------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Session already idle, `fresh=0` | return immediately, `immediate: true` | +| Session already idle, `fresh=1` | wait for the next transition into a requested state | +| Session dies mid-wait | resolve with `signal: "exit"` if `exit` was requested, otherwise resolve `timedOut:false, signal:null, ended:true`. Never hang | +| Session deleted mid-wait | same, resolve, do not throw. Verified live: `until=exit` gets `signal:"exit"`, a concurrent `until=blocked` gets `ended:true`, both in ~0ms | +| Shutdown with a wait pending | `cancelEverything()` in `stop()`. Verified live: SIGTERM with a 300s wait in flight exits in 1s | +| External CLI mode | `stop` and `blocked` never fire. Reject `until=stop` for those modes with a clear `INVALID_INPUT` rather than hanging until timeout | +| Multi-user | goes through `findSessionOrFail(ctx, id, req)`, which already enforces ownership | +| Remote / Docker cases | signals originate from the same `Session` object, so no special casing. Docker hooks need `CODEMAN_DOCKER_BRIDGE_HOOKS=1` for `stop`/`blocked` to arrive at all; without it, only `idle` works. Document it | +| Respawn `/clear` mid-wait | a respawn cycle emits `idle`. Callers waiting on `stop` are unaffected; callers on `idle` may resolve early. Documented, not fixed | +| Limit pause | if the session is paused on a usage limit, nothing will fire until the reset. The wait times out honestly. Consider surfacing `limitPaused: true` in the response so the caller can back off | + +### 3.7 Tests + +- `test/session-wait-registry.test.ts` (pure): immediate resolve, transition-required, multi-signal + first-wins, timeout, cap exceeded, cancel on session end, no listener leak after resolve, + chunk-straddling output match, case-insensitive match. +- `test/routes/session-wait-routes.test.ts` (`app.inject()`, no port): all three endpoints against + a `MockSession`, including the 200-with-`timedOut` contract and the ownership 404. +- `test/routes/session-input-wait.test.ts`: the send-and-wait race, plus proof that the non-wait + path is unchanged (still returns before `writeViaMux` settles). +- Live verification on a throwaway session before COM, per the always-end-to-end-test rule. + +### 3.8 Files touched + +- `src/config/agent-wait.ts` (new) +- `src/web/session-wait-registry.ts` (new) +- `src/web/session-listener-wiring.ts` (notify on idle/working/exit) +- `src/web/routes/hook-event-routes.ts` (notify on stop/blocked) +- `src/web/routes/session-routes.ts` (two new routes, `wait` fields on input) +- `src/web/schemas.ts` (`SessionWaitQuerySchema`, `SessionWaitOutputQuerySchema`, extend + `SessionInputWithLimitSchema`. Note: `.optional()` rejects `null`, so the frontend and any + generated client must send `undefined`, never `null`) +- `docs/api-reference.md`, `docs/extending-codeman.md`, README API table +- `skills/codeman/SKILL.md` recipes (Part 1 depends on this) + +--- + +## 4. Deferred: parts 3 to 5 + +Not in scope now, kept here so they are not lost. + +### Part 3: promote `blocked` to a first-class state + +`SessionStatus` is `'idle' | 'busy' | 'stopped' | 'error'`. "Needs you" exists three times over: +hook events, the `tab-alert-action` CSS class, and the phone overview NEEDS YOU section, each +re-deriving it. herdr makes `blocked` a real state that rolls up. + +Add `blocked` (and possibly `done`) to `SessionStatus`, set it from the same hook events that +Part 2 uses as wait signals, and clear it on the next `working`/`stop`. Then the tab strip, the +mobile overview, the wait endpoints, and any external agent read one field. + +Cost: `SessionStatus` is a widely-consumed union, so every exhaustive `switch` (the codebase has +`assertNever` and `noFallthroughCasesInSwitch`) will need a branch. That is a feature, it makes +the compiler find every site. This is a **minor** bump, not a patch: it widens a public type in +the HTTP contract. + +### Part 4: `GET /api/schema` + +herdr ships `herdr api schema`. Every Codeman route is already Zod-validated, so +`zod-to-json-schema` over `schemas.ts` gives a self-describing API almost free. Value: third-party +tools and the skill stop drifting from hand-written docs. Open question: whether to emit full +OpenAPI (`@fastify/swagger` would need per-route schema registration, which is a much larger +change) or just dump the Zod schemas keyed by name (cheap, 80% of the value). + +### Part 5: detection manifests instead of hardcoded patterns + +CLI-specific readiness, blocked and usage-limit patterns live in code across +`usage-limit-patterns.ts`, the respawn pattern helpers and `regex-patterns.ts`. Externalizing the +per-CLI ones into data files would make adding a sixth CLI a data change instead of a code change. + +**Do not copy the remote-update part.** herdr auto-fetches manifest updates from herdr.dev. +Codeman auto-pulling behavioral rules from a vendor server contradicts its security posture. +Bundled manifests plus local override only, no network. + +### Explicit non-goals + +- **Plugin runtime and marketplace.** `docs/extending-codeman.md` already argues this: a plugin + runtime means third-party code inside a process that spawns agents with your credentials, on a + server people expose over a tunnel. The reasoning still holds. If the marketplace _pattern_ is + wanted, apply it to data (web tabs, case templates, cron recipes), never to executable code. +- **Live PTY handoff on restart.** herdr needs it because it owns the terminals. Codeman + delegates to tmux, so PTYs already survive a self-update restart. +- **Socket API.** HTTP plus SSE is the existing, documented, stable contract. A second transport + would double the surface for no capability gain. + +--- + +## 5. Sequencing + +| Step | Work | Gate | +| ---- | ------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 1 ✅ | `src/config/agent-wait.ts` + `session-wait-registry.ts` + unit tests | 48 tests green | +| 2 ✅ | `GET .../wait` + wiring in listener-wiring, hook-event-routes, server teardown | 15 route tests green; live-verified on an isolated `CODEMAN_INSTANCE=waittest` instance (immediate resolve, 400 on a bad signal, 200+`timedOut` on timeout, hook `stop` and `permission_prompt`→`blocked` waking an in-flight wait, delete delivering `exit`, SIGTERM not blocked); full `test:ci` sweep green | +| 3 ✅ | `GET .../wait-output` | 16 route tests green; live-verified on real PTY bytes (`echo MARKER` waking a blocked request in ~1s, `from=buffer` immediate hit, never-seen marker timing out at exactly 2001ms, nocase, `regex` refused with a 400); full `test:ci` sweep green | +| 4 ✅ | `wait` field on `POST .../input`, non-wait path proven unchanged | 16 route tests green; live-verified (no-wait returns in 26ms with the historical bare body; an idle session did NOT satisfy a `wait` request, blocking the full 2001ms, which is the race the endpoint exists to close; the stop hook resolved a send-and-wait at 1510ms and the input was confirmed in the tmux pane; `wait:null` accepted) | +| 5 | `skills/codeman/SKILL.md` + reference files + `.claude/skills` symlink | live dogfood: a real session orchestrates a worker end to end | +| 6 | `codeman skill install` CLI + `applyAgentSkill()` + `agentSkillEnabled` setting | settings partial-PUT test, case-creation test | +| 7 | Docs: api-reference, extending-codeman, README | | +| 8 | COM (minor bump: new endpoints, new setting, new optional fields) | both CI and Release workflows green | + +Parts 1 and 2 are independent enough to land separately, but the skill is much less useful +without the wait endpoints, so the wait work goes first. + +## 6. Open questions for the owner + +1. `skills/` at the repo root, accepted despite the short-root rule? (Recommended yes, the + install one-liner depends on it.) +2. `agentSkillEnabled` default: OFF for the first release then flip, or ON immediately? +3. Auto-inject the skill into every case's `.claude/skills/`, or global install only? +4. Is `X-Codeman-Caller-Session` self-protection worth the 10 lines, given it is a footgun guard + and not a security boundary? +5. Regex support in `wait-output`: confirm literal-only for v1. + +--- + +## 7. Build log: what actually happened + +Written at the end of the build so the next person inherits the reasoning, not just the +diff. Process artifacts (per-agent briefs, findings, reports) live in the gitignored +`tmp/agent-wait-review/`; this section is the part worth keeping. + +### What shipped + +| Piece | Files | +| ------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- | +| Bounds + clamping | `src/config/agent-wait.ts` (new) | +| Blocking-wait registry | `src/web/session-wait-registry.ts` (new, IO-free, unit-tested) | +| `GET .../wait`, `GET .../wait-output`, `wait`/`waitTimeout` on `POST .../input` | `src/web/routes/session-routes.ts` | +| Signal wiring | `session-listener-wiring.ts` (idle/working/exit + output), `hook-event-routes.ts` (stop/blocked), `server.ts` (teardown, shutdown) | +| Agent skill | `skills/codeman/SKILL.md` + `reference/`, `.claude/skills/codeman` symlink, `package.json` `files` | +| Docs | `api-reference.md`, `extending-codeman.md`, `architecture-invariants.md`, `README.md`, `CLAUDE.md` | +| Tests | `test/session-wait-registry.test.ts`, three `test/routes/session-*wait*.test.ts`, `http-contract.test.ts`, `mock-session.ts` | + +### Bugs found in ADJACENT code, not in the new feature + +These are the highest-value output of the exercise and none were on the plan: + +1. **Every Codeman hook was dead on HTTPS installs.** `hooks-config.ts` built the hook + curl as `curl -s` with no `-k` while the statusline exporter 300 lines below used + `curl -sk` and documented why. Proven with the real hook command: `curl exit=60` + without the flag, success with it, and the failure swallowed by the hook's own + `2>/dev/null || true`. This silently killed `stop`, `permission_prompt`, + `elicitation_dialog`, `idle_prompt`, `teammate_idle` and `task_completed`, taking + respawn's definitive idle signals with them. Fixed, **plus** a staleness detector in + `refreshStaleCodemanHooks` that regenerates the on-disk config of already-created + cases (23 of 26 local cases carried the broken form; fixing the generator alone would + have left every one of them broken). +2. **`buildEnvExports()` exported a wrong-scheme `CODEMAN_API_URL`** (`http://` fallback + on an HTTPS install). Now omitted rather than guessed, so in-session guards fail closed. +3. **Programmatic input is only submitted when it contains `\r`.** `sendInput` sends Enter + only if the payload has a carriage return; without it the text sits in the composer + forever. Bit this build repeatedly before it was diagnosed, and had leaked into the + docs' own examples. + +### Design decisions worth not re-litigating + +- **A timeout is HTTP 200** with `wait.timedOut`, never a 4xx: callers loop over short + waits because tunnels cut idle connections, and every poll boundary would otherwise be + indistinguishable from failure. +- **Send-and-wait must be one endpoint.** A separate POST-then-wait races: between the + write and the flip to `working`, a wait sees the stale `idle` and reports the PREVIOUS + turn as this one. The waiter is registered before the write. +- **`stop`/`blocked` exist for `claude` mode only.** They come from Claude Code hooks; + `shell` installs none either, so keying off `isExternalCliMode()` was wrong. +- **Literal matching only, never regex.** JS `RegExp` backtracks; herdr can offer + `--regex` because Rust's regex crate is linear-time. +- **Client-hangup abort listens on `reply.raw` guarded by `writableFinished`.** On + `req.raw`, `close` fires when the request BODY ends, which on a POST killed every + send-and-wait instantly, and no `app.inject()` test can see it (inject never emits + `close`). +- **Liveness cannot come from `session.pid`.** For a tmux session that is the local + `tmux attach` client, not the worker: a worker exiting inside its pane leaves + `pane_dead=1` with the client alive, so `pid` never goes null. Liveness is probed at + the mux layer, cached (~750 ms) and only on blocking waits, never on the input hot path. + +### Verification rounds + +Six agents across three rounds, each verifying the previous round's work rather than its +own. Findings that mattered, in order of severity, were: the dead-pane liveness gap; the +`reply.raw` abort regression; abandoned long-polls leaking waiter slots; a crashed session +reporting `idle`; `shell` accepting `until=stop`; and a documented recipe that reported +success without running its task. Two traps recurred often enough to name: + +- **Vacuous passes.** `app.inject()` never emits `close`; a latched `cancelEverything()` + in `afterEach` silently killed the registry for every later test in a file; three test + files sharing one session id against the process-wide registry let one file's leftover + waiter fail another's assertion. Any new wait test needs care on all three. +- **HTTP-only test instances.** Every isolated instance used during the build was plain + HTTP, which is exactly why the HTTPS hook bug survived so long. Test the transport the + user actually runs. + +### Resolved at wrap-up (2026-08-08, conclusion pass) + +- **R2-A**: the fire-and-forget-then-gather-sequentially pattern was **removed from + the skill** rather than patched. Signals are edge-triggered with no history, so a + `stop` that fires before its waiter registers is unobservable afterwards; a + `fresh=0` gather was rejected because the only `until` set that current state can + satisfy answers `idle` for a prompt that never submitted, resurrecting the exact + false-success failure R2-B had just closed. Flow 3b's pattern B now gathers on + latched `wait-output` markers (`from=buffer`), the same mechanism that makes the + shell flows reliable; the limitation is recorded in + `architecture-invariants#agent-wait-primitives` and `endpoints.md`. The durable + fix, a latched last-signal-per-turn on the server, stays with deferred Part 3. +- Docs F7/F8, F4 and the false-`idle` attribution: `api-reference.md`, + `extending-codeman.md` and `architecture-invariants.md` rewritten to the post-fix + matcher (one normalized stream, chunk-straddling found, snippet as a rendering of + the matched window), the real no-PTY answer (`ended:true`, `aborted:false`, + `delivered:false`), and the startup-idle mechanism (a session parked on the trust + dialog emits no further `idle`; the false success is the startup transition). +- Orchestrate #12, #5/R2-B, #6, and R2-C..R2-E: fire-and-forget's empty `data` + documented; every send-and-wait retry loop now treats `duplicate:true` + + `immediate:true` as "no new turn ran" and reads the terminal before believing it; + claude fan-out is pattern A (backgrounded send-and-waits) or the marker gather; + readiness budgets rebalanced (5 s stage 1, 45 s stage 3) with the virgin-case + floor named; the auth fallback now also reads the supervisor definition + (`codeman-web.service` / launchd plist) and accepts `export`-prefixed `.env` + lines; `pid != null` is documented as startup-only, never liveness. +- Both public readiness recipes (extending-codeman.md, README) are bypass-first with + the trust probe as the bounded fallback; the worked recipe carries `-k` and fails + loudly on an empty SID; the hook `-k`/self-heal fix appears in every + "hooks go missing" list; the multi-word-TUI claim is "unreliable", not "never". + +### Still open + +- **Release checklist**: `package.json` `files` includes `skills`, which is still + untracked. `git add skills/` must be part of the release commit, or npm publishes + a tarball without the skill (a `files` entry that does not exist is silently + ignored, so nothing fails). +- The 1.13.0 changeset is written under `.changeset/`; consuming it (COM flow), + the release commit, and the deploy remain. +- Deferred with Part 3: the latched last-signal-per-turn. Nice-to-haves from the + reviews: N2 (create the death-watcher inside its `try`) and converting + timeout-shaped test detections into fast assertions. diff --git a/docs/api-reference.md b/docs/api-reference.md index 7866a898..957cf791 100644 --- a/docs/api-reference.md +++ b/docs/api-reference.md @@ -46,6 +46,20 @@ payload return `{ "success": true, "data": {} }`. > `GET /api/screenshots/:name`, `GET /q/:code` (QR redirect), and the > `GET /ws/sessions/:id/terminal` WebSocket upgrade. +> The [agent wait endpoints](#long-polling-agent-wait) use the normal envelope but +> are the only JSON endpoints that deliberately **hold the connection open**, for up +> to 600 s. Proxy operators and HTTP clients with a global read timeout need to know +> that before pointing them at Codeman. + +⚠️ **A `401` is the one status that is not an envelope.** Authentication is rejected +in a request hook, before any handler runs, and it replies with the bare string +`Unauthorized` (`Unauthorized: hook secret required` on the hook path) plus +`WWW-Authenticate: Basic realm="Codeman"`. There is no `success`, no `error`, and no +`errorCode`, because the wrapping hook only wraps object payloads. So a client that +pipes every response straight into a JSON parser dies with a parse error rather than +reporting an auth failure, which is a confusing way to discover that a password is +set. Branch on the HTTP status **before** parsing. + ## Error codes → HTTP status The single source of truth is `ErrorStatus` / `httpStatusForErrorCode()` in @@ -66,6 +80,333 @@ the HTTP status. Adding a new error code is non-breaking; removing or renaming one is a major change. +## Long-polling (agent wait) + +Three calls block until something happens instead of answering immediately. They +exist because SSE is Codeman's only other "tell me when" channel, and an agent +driving the API from a shell tool cannot practically hold a stream and parse +events inline. + +| Call | Blocks until | +|------|--------------| +| `GET /api/v1/sessions/:id/wait` | one of a set of lifecycle signals fires | +| `GET /api/v1/sessions/:id/wait-output` | a literal string appears in the session's output | +| `POST /api/v1/sessions/:id/input` with `wait` | the input is delivered **and then** a signal fires | + +`POST .../input` with `wait` is not the same as a `POST` followed by a separate +`GET .../wait`. It registers the waiter **before** writing, which closes the window +in which a separate wait sees the session still idle from the previous turn and +answers instantly with the wrong turn's result. Use it whenever you send a prompt +and want to know when that prompt is done. + +### Three semantics that break callers who assume otherwise + +**1. A timeout is HTTP `200`, not an error.** A wait that ends without its signal +returns `{"success":true, ...,"wait":{"timedOut":true,"signal":null}}`. The +intended pattern is a client-side loop over short waits, because `tailscale serve` +and cloudflared can both cut an idle connection, and turning every poll boundary +into a `4xx` would make that loop indistinguishable from a real failure. `408` is +auto-retried by several clients (silently doubling the polling load), `504` is what +a genuine tunnel failure looks like, and `204` cannot carry `waitedMs` / `status` / +`limitPaused`. Reserve error handling for the four codes in the table below. + +**2. `stop` and `blocked` fire only for `claude` sessions.** Both come from Claude +Code hooks, and no other mode installs them: `shell` runs no agent, and the external +CLIs (`opencode`, `codex`, `gemini`, `antigravity`) render their own TUIs and post +no hooks. For every non-`claude` mode only `idle`, `working` and `exit` are +accepted, and of those only `exit` is dependable: see the caveats under +[Signals](#signals) before building on `idle`. Requesting `stop` or `blocked` +**explicitly** on such a session is a +`400`; omitting `until` never fails, the server just drops them from the default set +and echoes the narrowed set back as `wait.until`. Three more places hooks can go +missing even in `claude` mode: a **Docker case** needs +`CODEMAN_DOCKER_BRIDGE_HOOKS=1`, since a container cannot reach a loopback-bound +Codeman (without it, only `idle` / `working` / `exit` work); a **remote-SSH +case** runs the agent on another host, whose hooks may never reach this server at +all; and a case whose hook config was written by **Codeman < 1.13.0 against an +`--https` install** carries hook curls without `-k`, which TLS-fail silently (the +hook line ends in `|| true`). Codeman now writes `curl -sk` and repairs a stale +case config the next time a session starts in that case. When in doubt, ask for +`stop,idle,exit` so a session without hooks still resolves on the heuristic +signal. + +**3. `from=now` does not mean "printed after you asked".** tmux repaints the visible +screen on attach, on resize, and on any TUI redraw, and a repaint arrives as +ordinary output, so text that was already on screen can satisfy a fresh wait. This +was observed live: a marker echoed a minute earlier matched instantly on a new +`from=now` wait. It is inherent to running the agent under a multiplexer, so the +contract is a **marker unique to each call** (`MARK="DONE_$RANDOM"`, send +`echo $MARK`, then wait on `$MARK`), never a generic string like `BUILD OK`. + +### Signals + +| Signal | Source | Actually fires for | +|--------|--------|--------------------| +| `idle` | the session's own `idle` event | `claude`: yes, on ❯-prompt detection after activity. `shell`: **once only**, ~500 ms after start, and never again. External CLIs: not guaranteed (they render their own TUIs and readiness is output stabilization) | +| `working` | the session's own `working` event | `claude` only in practice (spinner and work-keyword detection are Claude output formats) | +| `stop` | the Claude Code `stop` hook, the definitive end-of-turn signal | `claude` only | +| `blocked` | a `permission_prompt` or `elicitation_dialog` hook | `claude` only, and rarer than it looks: see below | +| `exit` | no process is behind the session | every mode | + +`stop` is the signal to orchestrate on where it exists; `idle` is a heuristic +fallback that can flap mid-turn when a spinner pauses. The default set when `until` +is omitted is `stop,idle,exit` (`exit` is in there so a worker that crashes resolves +the wait promptly instead of burning the caller's whole timeout on something that +can no longer happen). On a `claude` worker, prefer an explicit `until=stop,exit` +once the session is up: the default set's `idle` also resolves on a spinner pause, +and on a fresh session the **startup** `idle` (emitted when the CLI first comes up) +can land inside your first wait window and report a turn that never ran. Measured: +a session parked on the trust dialog emits no *further* `idle`, so it is the +startup transition, not the dialog, that produces the false success below. + +⚠️ **`exit` means "nothing is running", which includes "not started yet".** The +server answers from `pid === null` plus a mux-layer pane-death probe, and that +covers a session that exited — including a worker that died *inside* its tmux pane +while the local attach client (and therefore `pid`) lives on — one that was +detached, and one that was **created but never started**. So the first wait +after `POST /api/v1/sessions` returns `{"signal":"exit","immediate":true}` in +milliseconds, and reading that as "the worker died" is wrong: it means start it, or +wait for it to come up. `status` is carried alongside so nothing is hidden. The +alternative (trusting `status`) is worse, because a dead PTY parks the session at +`status: "idle"`, which would answer the default wait with `immediate: true` for a +worker that has crashed. A worker dying while a wait is parked resolves it within +a few seconds (a background death-watcher), not at the timeout. + +⚠️ **`blocked` is reachable less often than the table suggests.** It fires on two +hooks, and the default configuration suppresses one of them: Codeman spawns claude +with `--dangerously-skip-permissions`, so permission prompts do not happen unless the +instance is switched to the `auto` Claude mode (App Settings), or the caller is a +multi-user account without the bypass grant, which is forced to `--permission-mode +auto`. What does still fire under the default is `elicitation_dialog`, the agent +asking the user a question. So `until=stop,blocked,exit` is a reasonable belt on a +long turn, but a worker that never comes back is far more likely to be working than +blocked, and polling `blocked` alone will sit at its timeout. + +⚠️ **On a `shell` session, only `exit` and marker-matching are dependable.** A shell +session emits its one `idle` at startup and then stays `status: "idle"` forever, +whatever the pane is doing, so it never emits a *transition*. Since send-and-wait +requires a transition (and so does `fresh=1`), both can only time out there: +a documented default `wait` on a shell worker running `sleep 4` times out at the +full 25 s. Synchronize hook-less sessions with `wait-output` and a unique marker +instead. The same caution applies to the external CLIs. + +### Readiness is not a signal + +Nothing here reports "the agent is ready for a prompt", and no combination of +`until`/`fresh` synthesizes one. A freshly created session reads as `exit` (above), +and a `claude` worker in a brand-new case comes up on the CLI's **trust dialog**, +which contains a ❯ prompt of its own. Send-and-wait posted at that moment types the +prompt into the dialog, where the `\r` never gets past it, while the session's +startup `idle` lands inside the wait window: the wait resolves on `idle` in a +couple of seconds with `timedOut: false`, which looks exactly like a completed +turn. + +The reliable sequence is: poll `GET /api/v1/sessions/:id` until `.data.pid` is +non-null, then `wait-output` for the composer's own marker (`bypass`, the status +bar of a CLI spawned in bypass mode) with a short timeout, handling the trust +dialog only as the bounded fallback (`trust` matched → send `\r` → wait for +`bypass` again). Do not probe `trust` first and Enter blindly: the dialog text +stays in the terminal buffer for the life of the session, so a `trust` probe with +`from=buffer` keeps matching on every later run and the Enter lands in a ready +composer. A worked version is in +[`extending-codeman.md`](extending-codeman.md#seam-3-http-api-and-cli). + +### `GET /api/v1/sessions/:id/wait` + +| Param | Type | Default | Notes | +|-------|------|---------|-------| +| `until` | comma-separated list of `idle,working,stop,blocked,exit` | `stop,idle,exit` | resolves on the first to fire. An unknown token is a `400` naming it, never a silent fallback | +| `timeout` | positive integer ms | `60000` | **validated first, clamped second.** `0`, a negative value and a fractional value are all `400`s, not clamps; a valid value outside `[1000, 600000]` is clamped and echoed as `wait.timeoutMs` | +| `fresh` | `0` \| `1` \| `false` \| `true` | `0` | `1` requires an actual transition, ignoring the state at call time | + +```bash +curl -s "$API/api/v1/sessions/$SID/wait?until=stop,exit&timeout=60000" +``` + +Both GET wait routes answer with `Cache-Control: no-store`, because the documented +pattern polls one identical URL in a loop and a cached `{"timedOut":true}` would +turn that loop into a busy spin. `POST .../input` sends no cache header (it is a +POST, which is not heuristically cacheable). + +⚠️ **Unknown query parameters are ignored, not rejected**, with one exception +(`regex`, below). In particular `match=` on `/wait` is silently dropped and you get +a plain signal wait, so check the endpoint path before blaming the parameters. + +### `GET /api/v1/sessions/:id/wait-output` + +| Param | Type | Default | Notes | +|-------|------|---------|-------| +| `match` | literal string, 1 to 200 chars | required | substring match against the PTY stream with ANSI escapes stripped. A match spanning two PTY chunks is found | +| `nocase` | `0` \| `1` \| `false` \| `true` | `0` | case-insensitive compare. The returned snippet keeps the terminal's original casing | +| `from` | `now` \| `buffer` | `now` | `buffer` scans the tail of the existing terminal buffer (bounded, 256 KB by default) before blocking | +| `timeout` | positive integer ms | `60000` | same validation and clamp as `/wait` | + +**Matching is literal, never a pattern.** A `regex` parameter is rejected with a +`400` rather than ignored, so a caller that assumed otherwise finds out immediately +instead of waiting on the wrong thing. The reasoning is in +[`architecture-invariants.md`](architecture-invariants.md#agent-wait-primitives). + +#### What the matcher actually sees + +The matcher scans the raw PTY stream, **normalized**: ANSI escape sequences are +stripped — CSI, OSC, and the charset-designation escapes a stock bash prompt emits +on every line (`ESC ( B`), so `match=tnode:` matches a prompt that renders +`…@tnode:` — a partial escape arriving at a chunk boundary is held back until its +tail arrives, and a match may straddle PTY chunks: `printf STRAD; sleep 1; printf +DLEQQ` is matchable as `STRADDLEQQ` (all measured live). Three caveats remain: + +⚠️ **It is still the byte stream, not the rendered pane.** `GET .../terminal` +answers from a tmux screen capture (`data.source: "mux-visible"`), the finished +picture; the matcher sees the stream that painted it. For linear output the two +agree once escapes are stripped, but a full-screen TUI composes its picture with +cursor positioning, so what the pane shows and what the stream carries can differ. +Seeing your string in `terminal?tail=` makes a match likely, not guaranteed. + +⚠️ **A TUI's text can arrive without its spaces.** Claude Code positions words +with cursor moves rather than printing spaces, so screen text can reach the +matcher as `Quicksafetycheck:Isthisaprojectyoucreated...`. Whether a given phrase +keeps its spaces depends on how the TUI happened to draw it (measured: `I trust +this folder` matched, `Quick safety check` did not), so a multi-word `match` +against a TUI pane is unreliable rather than impossible. Match a **single +space-free token**, ideally one you printed yourself. Plain command output (a +shell worker, an `echo`) keeps its spaces. + +⚠️ **The returned `snippet` is a rendering of the matched text, not a quotation of +it.** It is cut from the same normalized stream the match ran against, then +cleaned for display: remaining raw control bytes are removed (an agent pipes the +snippet into its own terminal, so a worker's bytes must not be able to reset that +display) and blank runs are collapsed. A printable needle that matched will appear +in it; a needle containing control bytes or a blank run may not survive verbatim. + +```bash +MARK="DONE_$RANDOM" +curl -sG "$API/api/v1/sessions/$SID/wait-output" \ + --data-urlencode "match=$MARK" --data-urlencode 'timeout=120000' +``` + +Build the query with `-G --data-urlencode` rather than by hand: a `+` in a +hand-written query string decodes to a space. + +### `POST /api/v1/sessions/:id/input` with `wait` + +Two optional fields on the existing endpoint: + +| Field | Type | Notes | +|-------|------|-------| +| `wait` | `true` or the same comma grammar as `until` | `true` means the default signal set. Omitted keeps the historical fire-and-forget behavior, unchanged. `null`, `false` and an empty string are all read as **absent**, not as an error and not as "wait for the default" | +| `waitTimeout` | positive integer ms | same validation **and** clamp as `timeout`: `0`, a negative and a fractional value are `400`s, anything valid is clamped into `[1000, 600000]` and echoed as `wait.timeoutMs` | + +Both are `nullish`, so an explicit `null` from `JSON.stringify` is accepted as +"absent" rather than failing validation. That is deliberate: `.optional()` would +reject it, which has shipped as a real bug twice. + +The input must end with `\r` (a real carriage return in the JSON string): Enter is +sent only when the input contains one, so text without it is typed onto the +worker's prompt but never submitted, and the wait then runs its full timeout on a +turn that never started. Verified live; this is the most common silent failure on +this endpoint. + +```bash +curl -s -X POST "$API/api/v1/sessions/$SID/input" \ + -H 'Content-Type: application/json' \ + -d '{"input":"run the tests\r","useMux":true,"clientId":"agent-1","seq":1, + "wait":"stop","waitTimeout":600000}' +``` + +A **tagged duplicate** (a `clientId` + `seq` pair the server has already applied) +still honors `wait`, because the caller's question is unanswered, but it answers +from the session's current state rather than requiring a new transition: the +original turn may be long over. It comes back as +`"delivered": false, "duplicate": true`. + +### Response + +All three nest the wait result under `data.wait`, so one client helper works against +any of them: + +```json +{ "success": true, "data": { + "sessionId": "28325fd3-caa7-4178-82bf-87dfebf0f464", + "status": "idle", + "limitPaused": false, + "wait": { + "signal": "stop", "until": ["stop", "idle", "exit"], + "timedOut": false, "immediate": false, "ended": false, "aborted": false, + "waitedMs": 8421, "timeoutMs": 60000 + } +}} +``` + +`POST .../input` returns the same `wait` object alongside `delivered`, `duplicate`, +`status` and `limitPaused`. `POST .../input` **without** `wait` is unchanged and +still returns `{"success": true, "data": {}}`. + +⚠️ `delivered: false` has **two** meanings, and they must be told apart by +`duplicate`: with `duplicate: true` the input was suppressed as an already-applied +redelivery (harmless, the turn it refers to may be long over), while with +`duplicate: false` the **write failed** (typically no PTY behind the session). A +client that reads `delivered === false` as "duplicate" silently treats a failed send +as a success. + +| Field | Type | Meaning | +|-------|------|---------| +| `wait.signal` | signal \| `null` | the signal that fired (`/wait` and `/input` only) | +| `wait.until` | array of signals | what the server actually waited on, after narrowing the default set for the session's mode (`/wait` and `/input` only) | +| `wait.matched` | boolean | the string appeared (`/wait-output` only) | +| `wait.match` | string | the literal that was searched for (`/wait-output` only) | +| `wait.snippet` | string \| `null` | bounded window of output around the match, blank runs collapsed for readability (`/wait-output` only) | +| `wait.timedOut` | boolean | the wait hit its timeout. Still a `200` | +| `wait.immediate` | boolean | the condition already held at call time, so nothing was waited for (`waitedMs` is 0) | +| `wait.ended` | boolean | the session went away (deleted or torn down) before the condition was met | +| `wait.aborted` | boolean | the client hung up, so the waiter was released without resolving — and by that definition a client never reads `true`. When the **server** abandons a wait itself (send-and-wait against a session with no PTY), it answers in about a millisecond with `ended: true`, `delivered: false`, `duplicate: false` and `aborted: false`: `delivered`/`ended` carry that story, and `aborted` stays the transport flag. Present for completeness; treat a `true` as "this wait answered nothing", never as an outcome | +| `wait.waitedMs` | number | wall-clock ms actually spent waiting | +| `wait.timeoutMs` | number | the timeout **after clamping**, which is what was applied | +| `status` | `SessionStatus` | the session's status after the wait, so a caller that timed out still learns where things stand | +| `limitPaused` | boolean | the session is paused on a usage limit and will emit nothing until its reset, so a timeout here is expected rather than a stall worth retrying hard | + +Read the outcome by discriminator, in this order: + +1. `wait.signal !== null` (or `wait.matched === true`): the thing happened. +2. `wait.timedOut`: a poll boundary. Loop again. +3. `wait.ended` or `wait.aborted`: the wait answered nothing, because the session is + gone or was never running. Re-check the session instead of looping. + +`wait.immediate` is not a fourth outcome: it rides along with the first one and +means the condition already held at call time, so nothing was actually waited for. +If that is not what you meant, you wanted `fresh=1` or the send-and-wait form. Note +that `{"signal":"exit","immediate":true}` on a session you just created is the +not-started-yet case, not a crash. + +**The timeout is clamped, so read it back.** A request for 1800000 ms is silently +reduced to the server's ceiling (600000 ms by default, operator-tunable), and a +request for 1 ms is raised to 1000 ms. `wait.timeoutMs` is the value that was +applied. Without checking it, a caller that asked for 30 minutes and got 10 will +read the timeout as "the worker is wedged" and kill a session that was working fine. + +### Errors + +| `errorCode` | HTTP | When | +|-------------|------|------| +| `INVALID_INPUT` | 400 | unknown `until` / `wait` token; `stop` or `blocked` requested explicitly on a mode that installs no hooks (the message names the mode); `regex=` on `/wait-output`; `match` outside 1 to 200 chars; a non-numeric `timeout` | +| `NOT_FOUND` | 404 | no such session, or one this caller does not own | +| `SESSION_BUSY` | 409 | this session's waiter cap is full | +| `RATE_LIMITED` | 429 | a per-owner or process-wide waiter cap is full. Retry later; the session you named is not the problem | + +The two capacity codes are deliberately different. A process-wide cap reported as +`SESSION_BUSY` would tell the caller to switch sessions, which cannot help. The +error message names the cap that was hit. + +⚠️ A `401` is **not** in this table and is not an envelope at all (see +[Response envelope](#response-envelope)). It matters most here: a polling loop that +pipes each wait straight into `jq` fails with a parse error on every iteration +against a password-protected server, which reads as "the wait endpoints are broken". +Check the status first. + +The per-session cap is a **combined** budget: signal waiters and output waiters +count against the same 16, not 16 of each. An abandoned request no longer holds its +slot, because the routes release the waiter when the client disconnects, but a +client that opens many concurrent waits against one session will still hit the cap. + ## Authentication Optional HTTP Basic (`CODEMAN_USERNAME`/`CODEMAN_PASSWORD`) → opaque diff --git a/docs/architecture-invariants.md b/docs/architecture-invariants.md index 040bdaf2..c865c431 100644 --- a/docs/architecture-invariants.md +++ b/docs/architecture-invariants.md @@ -42,6 +42,20 @@ Implementation detail extracted from `CLAUDE.md` so that file stays small enough **Input**: `session.writeViaMux()` for programmatic/curl input — tmux `send-keys -l` (literal) + `send-keys Enter`. Single-line only (fire-and-once). Interactive **browser** input goes through a durable **exactly-once** layer: each frame carries a stable `clientId` + monotonic per-session `seq`, persisted to localStorage until the server ACKs (`{t:'ia',seq}` over WS, or HTTP 2xx), so a dropped link/reconnect can't lose or double-deliver a prompt. **WS resilience** (#149): the upgrade URL carries `cid = clientId + ':' + perTabNonce`, and `ws-connection-registry.ts` supersedes only same-TAB reconnects (two tabs on one session coexist; input frames keep the bare `clientId` for seq dedup); reconnects back off exponentially (attempts preserved across `_connectWs`), and the header connection chip renders from a real `_wsState` lifecycle (`connecting`/`connected`/`fallback`/`reconnecting`/`disconnected`). +### Agent wait primitives + +**Agent wait primitives** (`GET /api/sessions/:id/wait`, `GET /api/sessions/:id/wait-output`, and the `wait`/`waitTimeout` fields on `POST /api/sessions/:id/input`): bounded long-polls that let an agent driving Codeman from a shell tool block until something happens. They exist because SSE was the only "tell me when" channel Codeman had, and a curl-driven caller cannot practically hold a stream and parse events inline. The blocking core is `src/web/session-wait-registry.ts` (no IO, no `Session` reference, so it unit-tests in isolation), bounds live in `src/config/agent-wait.ts`, and the wiring is three `notifySignal()` calls next to existing broadcasts (`session-listener-wiring.ts` for `working`/`idle`/`exit`, `hook-event-routes.ts` for `stop`/`blocked`) plus `notifyOutput()` riding the already-attached `terminal` listener. Design: `docs/agent-control-plan.md` §3; wire contract: `docs/api-reference.md`. + +**Ordering rules, both load-bearing and both invisible to a reader of either side alone.** ⚠️ **exit-before-cancel**: `_doCleanupSession` in `server.ts` must call `sessionWaits.notifySignal(id, 'exit')` BEFORE `sessionWaits.cancelAll(id)`, because that method detaches the session's listeners before `session.stop()`, so on a delete the PTY's own exit event never reaches the registry and an `until=exit` caller would get a bare `ended: true` instead of the signal it asked for. Found by live-testing the delete path, not by the unit tests. ⚠️ **registered-before-write**: the send-and-wait path on `POST .../input` registers the waiter BEFORE writing to the PTY, and that ordering is the entire reason the combined endpoint exists rather than documenting "POST, then GET .../wait": between the write and the session flipping to `working` there is a window in which a separate wait sees the session still idle and instantly reports the PREVIOUS turn as this turn's answer. Two consequences hang off it: `useMux` delivery is awaited on that path (the response is staying open anyway, so a `writeViaMux` failure becomes observable for the first time) while the non-wait path keeps its fire-and-forget shape byte for byte, and because `shouldApplyInput()` MUTATES (it records the seq) before registration, a registration that fails on a full pool must `forgetInputSeq()` before returning, or the caller's retry is rejected as a duplicate and the input is lost by the very mechanism reliable delivery exists for. + +**A timeout is a 200, deliberately.** `{"timedOut": true, "signal": null}` with HTTP 200 is the long-poll succeeding at answering "did this happen within N ms?" with "no". The documented client pattern is a loop over short waits (`DEFAULT_WAIT_MS` is 60s precisely because prod is reached through `tailscale serve` and users run cloudflared, both of which cut idle connections), and turning every poll boundary into a 4xx would make that loop indistinguishable from a real failure. The alternatives are all worse: `408` is auto-retried by several clients and proxies, silently doubling the polling load; `504` is what a genuine tunnel failure looks like, so reusing it destroys the caller's ability to tell the two apart; `204` cannot carry `waitedMs`/`status`/`limitPaused` and breaks the uniform envelope the versioning policy makes a stable promise. Errors are reserved for `INVALID_INPUT` (400), `NOT_FOUND` (404, also the multi-user ownership answer via `findSessionOrFail`), `SESSION_BUSY` (409, this session's cap) and `RATE_LIMITED` (429, the per-owner or process-wide cap: a global cap reported as `SESSION_BUSY` tells the caller to switch sessions, which cannot help). ⚠️ The clamp is silent, so the EFFECTIVE timeout is echoed back as `wait.timeoutMs`: a caller that asked for 30 minutes, got the 600s ceiling, and could not see it would read the timeout as "the worker is wedged" and kill a session that was working fine. All three endpoints nest the result under `data.wait` for the same reason, so one client helper works against any of them instead of an agent's `is_done()` reading `undefined` off the shape it did not expect. + +**Matching is literal, and that is a language constraint, not a missing feature.** `search-service.ts` already avoids regex so there is no ReDoS surface, and this endpoint is more exposed still: the pattern is caller-supplied and the input is a live stream. herdr can offer `--regex` on its equivalent because Rust's regex crate is linear-time with no backtracking; JavaScript's `RegExp` backtracks, so the same feature here is a denial-of-service primitive. A `regex` parameter is therefore REJECTED with a 400 rather than ignored, since an agent that assumed otherwise would silently wait on the wrong thing. `match` is capped at 200 chars (`MAX_MATCH_LENGTH`), which is also what keeps the per-waiter carry buffer small: a match can straddle two PTY chunks, so each waiter carries `match.length - 1` characters of the previous chunk and tests `carry + chunk`. ⚠️ **`from=now` does not mean "printed after you asked"**: tmux repaints the visible screen on attach, on resize, and on any TUI redraw, and a repaint arrives as ordinary `terminal` data, so text already on screen can satisfy a fresh wait (observed live: a marker echoed a minute earlier matched instantly). This is inherent to running the agent under a multiplexer and is not fixable in the registry, so the contract is a marker unique per call (`echo DONE_$RANDOM`), never a generic one like `BUILD OK`, and every recipe must show that. ⚠️ **There is exactly ONE definition of the matched stream, `normalizeForMatch()`**: `stripAnsi()` (CSI, OSC, `ESC =`/`ESC >`) plus `ANSI_ESCAPE_RESIDUE` for what that helper leaves behind — above all the `ESC ( B` charset switch a stock bash prompt emits on every line, which in the first build survived into the matched text and made `match=tnode:` silently fail against a prompt that plainly renders `tnode:`. The haystack, the carry and the snippet window all derive from that one function, so matching and the snippet cannot drift apart again. ⚠️ **`splitTrailingEscape()` holds back a partial escape at a chunk boundary** until its tail arrives; its `INCOMPLETE_ANSI_TAIL` pattern must stay in lockstep with `ANSI_ESCAPE_RESIDUE` (a chunk cut between the `(` and the `B` is otherwise a fresh way to smuggle an escape into the haystack), and it is deliberately non-global (the repo-wide `lastIndex` hazard). With the carry, a match may straddle PTY chunks: `printf STRAD; sleep 1; printf DLEQQ` is matchable as `STRADDLEQQ` (measured live, both `from` modes). ⚠️ **The matcher still sees the byte stream, not the rendered pane** (`GET .../terminal` is a tmux capture, `source:'mux-visible'`): linear output agrees once escapes are stripped, but Claude Code positions words with cursor moves instead of spaces, so TUI text can arrive space-less (`Quicksafetycheck:Isthis...` — measured; some phrases keep their spaces depending on how the TUI drew them), which is why the documented advice remains ONE short space-free token the caller printed itself. The returned `snippet` is a RENDERING of the matched window, not a quotation: `SNIPPET_CONTROL_BYTES` drops bare control bytes that carry no ESC (BEL, NUL, backspace — `normalizeForMatch` removes escape SEQUENCES only) and blank runs are collapsed, because the consumer is an agent piping it through `jq` into its OWN pane, where a worker's raw bytes could otherwise reset or garble the orchestrator's display. + +**Lifetime discipline, per the 24-hour-session rules.** Every waiter owns exactly one timer, cleared on resolve; per-session waiter sets are deleted when they empty; `cancelAll()` runs on session teardown and `cancelEverything()` in `stop()`. ⚠️ **Waiter timers are deliberately NOT unref'd**, the opposite of the usual advice: an unref'd timer would let the process exit mid-wait and strand the HTTP response, so shutdown resolves waiters explicitly instead. All three routes also release the waiter when the client disconnects (`abortOnClientHangUp()` in `session-routes.ts`), which the caps make load-bearing rather than tidy: the documented loop-over-short-waits pattern is naturally written as `curl --max-time 30 ".../wait?timeout=60000"`, and without it every iteration abandons a waiter that lives out its full timeout, so the seventeenth call gets a 409 for a session nobody else is waiting on. ⚠️ **That listener goes on `reply.raw`, guarded by `writableFinished`, NOT on `req.raw`** (the obvious choice, and the one the SSE route in `server.ts` can afford because it only ever serves a GET). `req.raw` emits `close` as soon as the REQUEST BODY has finished streaming, which on a POST happens before the handler blocks: measured at +1ms with `aborted: false`, indistinguishable from a real hang-up, so wiring it there cancels every send-and-wait instantly and silently kills the feature, while GET keeps working because a GET has no body to finish. `reply.raw` emits `close` both on a completed response and on a dead socket, and `writableFinished` is the only thing that separates them, so the guard is load-bearing rather than defensive. ⚠️ `app.inject()` never emits `close` at all, so none of this is observable in a route test: the regression test has to bind a real port. `MAX_WAITERS_PER_SESSION` (16) is a COMBINED signal-plus-output budget (`waiterCount()` sums both maps), not 16 of each; `MAX_WAITERS_PER_OWNER` (48) applies only when an owner is passed, so single-user mode behaves exactly as before; `MAX_WAITERS_TOTAL` (128) mirrors `MAX_SSE_CLIENTS` in `map-limits.ts`, since each pending waiter costs an open HTTP response plus a timer. Capacity is asserted BEFORE the expensive work on both sides: `waitForOutput()` checks before the `initialText` scan, and the `wait-output` route checks before reading `session.terminalBuffer`, whose getter joins the whole 32MB accumulator, so a request that is going to be rejected never pays for a buffer materialization. `from=buffer` scans only the tail (`MAX_BUFFER_SCAN_BYTES`, 256KB) because the question it answers is "did this appear recently", not "ever", and the tail is continuous with the live stream (`append()` and `emit('terminal')` receive the same bytes). + +**Signal availability is decided by MODE, not by `isExternalCliMode()`.** `stop` and `blocked` come from Claude Code hooks, so `hooksAvailableForMode()` is true only for `claude`: `shell` is not an external-CLI mode but installs no hooks either, so `until=stop` on a shell session is a guaranteed unresolvable wait dressed up as a timeout. The behavior split is deliberate and must survive refactors: an EXPLICIT request for an unavailable signal is a 400 naming the mode, while the DEFAULT set (`stop,idle,exit`) silently drops them and echoes the narrowed set back as `wait.until`, because omitting the parameter must never 400. Signal quality is not uniform either: `stop` is definitive (Claude Code says the turn is over), `idle` is inferred from output stabilization plus prompt detection and can flap mid-turn when a spinner pauses, which is why `stop` is the documented default to orchestrate on and `idle` is the fallback for sessions that emit no hooks. ⚠️ **`idle` being ACCEPTED for a mode does not mean it ever FIRES there.** `startShell()` emits exactly one `idle` on a 500ms readiness timer and nothing afterwards, so a shell session sits at `status:'idle'` no matter what its pane is doing; since send-and-wait and `fresh=1` both require a TRANSITION, both can only time out on a shell worker (measured: a default `wait` on `sleep 4` burned its full 25s). The ❯-prompt and spinner detection that drives the real `working`/`idle` cycle is Claude's output format, so hook-less modes synchronize with `wait-output` markers or `exit`, and the docs must say so rather than listing `idle` as "available" and letting the reader infer it is usable. ⚠️ **Signals are edge-triggered with no history**, and that is a real orchestration limit: a signal that fires while no waiter is registered is gone, unobservable by any later wait variant (`until=stop` after the turn ended just times out, `fresh` or not — measured, R2-A). Documented client patterns must therefore register the waiter before the event can fire (send-and-wait) or gather on latched `wait-output` markers with `from=buffer`; the skill's fan-out flow was rewritten accordingly, and "fire-and-forget N prompts, then gather signal-waits sequentially" must never be documented again. The durable fix, a server-side latched last-signal-per-turn, is deferred with Part 3 of `docs/agent-control-plan.md`. ⚠️ Relatedly, the route corrects liveness that `SessionStatus` cannot express: `currentSignalFor()` answers `exit` whenever `pid === null` (exited, detached, or created-and-never-started) **or the mux pane is dead**, because `Session` parks a DEAD PTY at `_status = 'idle'` and trusting the status would answer the default wait `{signal:'idle', immediate:true}` for a crashed worker while `until=exit` blocked forever on an event that already happened. ⚠️ **`pid` alone cannot carry liveness for a tmux-backed session**: that pid is the local `tmux attach` CLIENT, so a worker exiting inside its pane leaves `pane_dead=1` with the client alive and `pid` never goes null — the `pid === null` branch is unreachable in the normal configuration (unit tests exercise it because `MockSession` sets `pid` by hand; only a live instance showed the gap). Liveness is therefore probed at the mux layer: `workerIsDead()` consults `mux.isPaneDead(muxName)` with a ~750 ms per-pane cache, ONLY on blocking waits (measured: 0 tmux execs across 100 non-wait input POSTs — the browser hot path pays nothing), plus a refcounted 3 s `watchForDeadWorker` interval so a worker dying while a wait is parked resolves it in ~3 s instead of burning the timeout. The probe fails SAFE by construction ("cannot tell" is never "dead": non-mux sessions, a missing or throwing `isPaneDead`, all return false). On send-and-wait, a "successful" `send-keys` into a dead pane additionally overrides `delivered` to `false` and rolls the dedup seq back (`undoOnFailure`), because the bytes went nowhere and a retry against a restarted worker must not be refused as a duplicate. The cost is that a just-created session reads as `exit`, which the wire docs must spell out as "not started yet"; the fix lives at the route rather than in `signalForStatus()` because the registry deliberately holds no `Session` reference. ⚠️ Two `claude`-mode cases still lose hooks for reasons outside the registry: a Docker case cannot reach a loopback-bound Codeman without `CODEMAN_DOCKER_BRIDGE_HOOKS=1`, and a remote-SSH case runs the agent on another host whose hooks may never reach this server. Bounds are env-overridable (`CODEMAN_WAIT_MAX_MS`, `CODEMAN_WAIT_DEFAULT_MS`, `CODEMAN_WAIT_MAX_PER_SESSION`, `CODEMAN_WAIT_MAX_PER_OWNER`, `CODEMAN_WAIT_MAX_TOTAL`, `CODEMAN_WAIT_BUFFER_SCAN_BYTES`) and each is clamped to a hard bound so a typo degrades to the default instead of disabling the protection; they are internal tuning knobs like the rest of `src/config/`, NOT part of the SemVer-covered env-var surface in `versioning-policy.md`. Tests: `test/session-wait-registry.test.ts`, `test/routes/session-wait-routes.test.ts`, `test/routes/session-wait-output-routes.test.ts`, `test/routes/session-input-wait.test.ts`. + ### Auto-resume on usage limit **Auto-resume on usage limit** ("token pause" control, opt-in per session, top of the Respawn tab): when Claude halts on a subscription limit ("5-hour limit reached ∙ resets 8pm" and all 1.0.x–2.1.x variants), `usage-limit-patterns.ts` (pure, unit-tested) parses the reset time from cleaned output; `SessionAutoOps` arms a timer for reset+2min, then sends Esc (dismisses the rate-limit dialog) + `continue`. Still-limited responses re-arm the loop (5-min retry on stale times); a `working` transition cancels it. Claude-mode only (detection rides `_processExpensiveParsers`). Persists/recovers via `SessionState.autoResumeEnabled`/`autoResumeAt`; respawn cycles are blocked while paused (`isLimitPaused` guard in `onIdleDetected` — prevents `/clear` from wiping the paused conversation). Endpoint: `POST /api/sessions/:id/auto-resume`; SSE: `session:limitPauseScheduled`/`limitResume`/`limitResumeCancelled`. Tests: `test/usage-limit-patterns.test.ts`, `test/session-auto-resume.test.ts`. diff --git a/docs/extending-codeman.md b/docs/extending-codeman.md index 5d9de299..c68a2f46 100644 --- a/docs/extending-codeman.md +++ b/docs/extending-codeman.md @@ -44,6 +44,10 @@ is in [`api-reference.md`](api-reference.md). the payload at the top level rather than under `data`. Read defensively with `body.data ?? body`. +⚠️ A `401` is not an envelope at all: auth is rejected in a request hook that +replies with the bare string `Unauthorized`, so parsing it as JSON throws. Branch on +the status code before you parse, or a missing password looks like a broken endpoint. + **Already driving Codeman from an agent?** The README's [Programmatic Guide](../README.md#driving-codeman-from-an-agent--programmatic-guide) covers the in-session case: the `CODEMAN_MUX`, `CODEMAN_API_URL`, @@ -152,7 +156,7 @@ for (;;) { ## Seam 3: HTTP API and CLI -Around 199 handlers across 21 route files cover sessions, cases, files, cron, +Around 200 handlers across 21 route files cover sessions, cases, files, cron, respawn, Ralph, the orchestrator, search, and admin. Each route module carries an `@fileoverview` describing its endpoints. @@ -167,10 +171,12 @@ curl -u admin:$PASS -X POST http://127.0.0.1:3000/api/v1/sessions \ -H 'Content-Type: application/json' \ -d '{"workingDir":"/home/me/project","mode":"claude"}' -# Send a prompt (single-line only) +# Send a prompt (single-line only, and it must end with \r: Enter is sent only +# when the input contains a carriage return; without it the text sits on the +# session's prompt unsubmitted) curl -u admin:$PASS -X POST http://127.0.0.1:3000/api/v1/sessions/$ID/input \ -H 'Content-Type: application/json' \ - -d '{"input":"run the tests","useMux":true}' + -d '{"input":"run the tests\r","useMux":true}' ``` `POST .../input` also accepts `clientId` (stable per client, max 128 chars) and @@ -178,6 +184,124 @@ curl -u admin:$PASS -X POST http://127.0.0.1:3000/api/v1/sessions/$ID/input \ at-most-once, so retrying after a dropped connection cannot type the prompt twice. Omit them entirely rather than sending `null`. +It also accepts `wait` and `waitTimeout`, which hold the response open until the +session finishes the turn you just started. `wait` is `true` (the default signal +set) or a comma list of `idle,working,stop,blocked,exit`; the result comes back +under `data.wait`. Sending them changes nothing for callers that do not: without +`wait` the response is still `{"success": true, "data": {}}` and the write is still +fire-and-forget. The two interact with `clientId` / `seq` in one way worth knowing: +a **tagged duplicate** (a pair the server already applied) skips the write but still +waits, answering from the session's current state rather than blocking for a +transition that already happened. It reports `"delivered": false, "duplicate": true`. + +### Waiting instead of polling + +Three calls block until something happens: `GET /api/v1/sessions/:id/wait` (a +lifecycle signal), `GET /api/v1/sessions/:id/wait-output` (a literal string in the +output), and the `wait` field above. Full parameter and response tables are in +[`api-reference.md`](api-reference.md#long-polling-agent-wait). Four things decide +whether your integration works, and the last one is what actually bites: + +- **A timeout is a `200` with `wait.timedOut: true`**, not an error. Loop over short + waits rather than issuing one long one, because `tailscale serve` and cloudflared + both cut idle connections and a single 10-minute call is the pattern most likely + to die in the field. +- **`wait.timeoutMs`** is the timeout after server-side clamping (600 s ceiling by + default). Read it rather than assuming you got what you asked for. +- **`stop` and `blocked` only exist for `claude` sessions**, and on a `shell` session + even `idle` fires only once at startup, so send-and-wait there can only time out. + See the Gotchas below. + +⚠️ **There is no readiness signal, and skipping readiness is the failure that looks +like success.** A session reports `idle` before its CLI has spawned, and a `claude` +worker in a brand-new case comes up on the CLI's **trust dialog**, which has a ❯ +prompt of its own. Prompt it at that moment and the text lands in the dialog, the +`\r` does not get past it, and the session's startup `idle` lands inside the wait +window: the wait resolves on `idle` in a couple of seconds with `timedOut: false`, +indistinguishable from a finished turn. Wait for the pid, then wait for the +composer, answering the dialog only as the bounded fallback. + +A worked orchestration: start a worker, get it ready, prompt it, wait, clean up. + +```bash +API="${CODEMAN_API_URL:-http://127.0.0.1:3000}" # auto-set in-session, correct scheme included +AUTH=(-u "admin:$CODEMAN_PASSWORD") # omit entirely if no password is set +CURL=(curl -sk "${AUTH[@]}") # -k: harmless on http, required on --https installs (self-signed cert) + +# 1. Start a worker session (creates the case if it does not exist yet). +# The guard matters: a TLS or auth failure otherwise leaves SID empty and every +# later step "succeeds" against nothing. +SID=$("${CURL[@]}" -X POST "$API/api/v1/quick-start" \ + -H 'Content-Type: application/json' \ + -d '{"caseName":"worker-1","mode":"claude"}' | jq -r '.data.sessionId') +[ -n "$SID" ] && [ "$SID" != null ] || { echo "quick-start failed"; exit 1; } + +# 2. READINESS: composer marker first, trust dialog only as the bounded fallback. +# Skip this and step 3 reports a turn that never ran. Do NOT probe trust first +# and Enter blindly: the dialog text stays in the buffer for the life of the +# session, so on every later run that probe matches stale text and the Enter +# lands in a ready composer. Match single tokens only: TUI text can arrive +# without its spaces. Stage 1 is short on purpose (an already-trusted case +# matches in <1 s; a first-run case can never pass it and pays it in full). +until [ "$("${CURL[@]}" "$API/api/v1/sessions/$SID" | jq '.data.pid')" != null ] +do sleep 1; done +R=$("${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \ + --data-urlencode 'match=bypass' --data-urlencode 'from=buffer' \ + --data-urlencode 'timeout=5000') # composer's status bar = ready +if ! jq -e '.data.wait.matched' <<<"$R" >/dev/null; then + T=$("${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \ + --data-urlencode 'match=trust' --data-urlencode 'from=buffer' \ + --data-urlencode 'timeout=2000') + jq -e '.data.wait.matched' <<<"$T" >/dev/null && \ + "${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" \ + -H 'Content-Type: application/json' -d '{"input":"\r","useMux":true}' >/dev/null + "${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \ + --data-urlencode 'match=bypass' --data-urlencode 'from=buffer' \ + --data-urlencode 'timeout=45000' >/dev/null +fi + +# 3. Send the prompt AND register the wait in one call, so the answer cannot be +# the previous turn's idle state. Single line only, ending in \r (otherwise +# Enter is never sent and this wait times out on a turn that never started). +W=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" \ + -H 'Content-Type: application/json' \ + -d '{"input":"Run the test suite and summarize the failures\r","useMux":true, + "clientId":"orchestrator","seq":1,"wait":"stop,exit","waitTimeout":60000}' \ + | jq -c '.data.wait') + +# 4. That first wait probably timed out (60 s). Keep going in SHORT waits. +for _ in $(seq 1 30); do + [ "$(jq -r '.timedOut' <<<"$W")" = 'true' ] || break # signal fired, or wait ended + W=$("${CURL[@]}" \ + "$API/api/v1/sessions/$SID/wait?until=stop,exit&timeout=60000" | jq -c '.data.wait') +done +jq -r 'if .ended or .aborted then "worker is not running" + elif .timedOut then "still working after 30 waits" + else "signal: \(.signal)" end' <<<"$W" + +# 5. Read what it produced, then delete the session YOU created, by exact id. +# ⚠️ NOT /output: its textOutput is empty for every tmux-backed session. +# `tail` counts BYTES, and the payload is terminal data with ANSI in it. +"${CURL[@]}" "$API/api/v1/sessions/$SID/terminal?tail=8000" | jq -r '.data.terminalBuffer' +"${CURL[@]}" -X DELETE "$API/api/v1/sessions/$SID" +``` + +Waiting on a marker instead of a signal is the form that works in **every** mode, +and the only one that works on a `shell` session: + +```bash +# ⚠️ Split the marker so the typed line never contains it: your own keystrokes echo +# into the output stream, so an unsplit marker matches before the command has run. +# `from=buffer` also catches a marker that printed before the wait registered. +N=$RANDOM +"${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" \ + -H 'Content-Type: application/json' \ + -d "{\"input\":\"M=DONE; npm test; echo \${M}_$N rc=\$?\r\",\"useMux\":true}" +"${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \ + --data-urlencode "match=DONE_$N" --data-urlencode 'from=buffer' \ + --data-urlencode 'timeout=60000' | jq '.data.wait' +``` + For shell scripting, the `codeman` CLI is the same surface without the HTTP plumbing: @@ -221,10 +345,41 @@ Every one of these has cost somebody real time. shipped bugs more than once. - **`text/plain` bodies stay raw.** Auto-parsing them as JSON enabled simple-request CSRF, so it is deliberate. Send `application/json`. -- **Prompts are single-line.** With `useMux: true` the server delivers your text - and then Enter as two separate writes, so you do not append `\r` yourself. A - multi-line string breaks the agent's Ink-based input handling: send one line, - or split it across calls. +- **Prompts are single-line and must end with `\r`.** The server splits your text + and Enter into two separate tmux writes (Ink needs them apart), but it sends the + Enter **only when the input contains a carriage return**. Without it your text + sits on the prompt unsubmitted, which is the single most common "the wait + endpoints don't work" report: the wait runs its full timeout on a turn that never + started. Newlines inside the string are stripped rather than rejected, so + `"echo A\necho B\r"` runs the single joined command `echo Aecho B`: send one line + per call. +- **`wait-output`'s `from=now` is not "printed after you asked".** tmux repaints + the visible screen on attach, on resize, and on any TUI redraw, and a repaint + arrives as ordinary output, so text already on screen can satisfy a fresh wait. + Observed live: a marker echoed a minute earlier matched instantly. Use a marker + unique to each call, and build it so the typed line never contains it (your own + keystrokes echo into the stream). Matching is a literal substring, so `regex=` is + rejected with a `400` rather than ignored. +- **`wait-output` matches the normalized PTY stream, not the screen.** ANSI escape + sequences are stripped (the `ESC ( B` charset escape a bash prompt emits on every + line included), a partial escape at a chunk boundary is held back until its tail + arrives, and a match may straddle PTY chunks, so text you printed yourself + matches reliably (`printf STRAD; sleep 1; printf DLEQQ` is matchable as + `STRADDLEQQ`). What can still fail is TUI output: a full-screen TUI positions + words with cursor moves, so its text can reach the matcher **without spaces** and + a multi-word match is unreliable there. Match one short space-free token, ideally + one you printed yourself, and keep it out of the typed line (your own keystrokes + echo into the stream). +- **`stop` and `blocked` never fire for `shell`, `opencode`, `codex`, `gemini` or + `antigravity` sessions.** They come from Claude Code hooks, which no other mode + installs, so only `idle`, `working` and `exit` exist there. Asking for them + explicitly is a `400`; omitting `until` is safe, since the server drops them from + the default set and echoes what it actually waited on as `wait.until`. Even in + `claude` mode, a Docker case needs `CODEMAN_DOCKER_BRIDGE_HOOKS=1` for hooks to + reach the server at all, a remote-SSH case's hooks may never arrive, and a case + written by Codeman < 1.13.0 against an `--https` install carries hook curls + without `-k` that TLS-fail silently — a 1.13.0+ server rewrites them the next + time a session starts in that case. - **Unwrap the envelope** before reading fields. `data` is not the response body. ## Publishing your integration diff --git a/package-lock.json b/package-lock.json index ce72d462..6846c328 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "aicodeman", - "version": "1.12.2", + "version": "1.13.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "aicodeman", - "version": "1.12.2", + "version": "1.13.0", "hasInstallScript": true, "license": "MIT", "workspaces": [ diff --git a/package.json b/package.json index 263bc3b9..01502459 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "aicodeman", - "version": "1.12.2", + "version": "1.13.0", "description": "Mission control for AI coding agents - run 20 autonomous agents with real-time monitoring and session persistence", "type": "module", "main": "dist/index.js", @@ -159,6 +159,7 @@ "dist", "scripts/postinstall.js", "scripts/fix-node-pty.mjs", + "skills", "LICENSE", "README.md" ] diff --git a/skills/codeman/SKILL.md b/skills/codeman/SKILL.md new file mode 100644 index 00000000..7ff25f68 --- /dev/null +++ b/skills/codeman/SKILL.md @@ -0,0 +1,274 @@ +--- +name: codeman +description: >- + Drive Codeman, the session manager this agent is running inside, over its HTTP API: + list sessions, start worker sessions, send them prompts, block until they finish + (wait / wait-output / send-and-wait), read their output, and clean up. Use when asked + to orchestrate or parallelize work across Codeman sessions, watch another session, or + start and manage workers. Only usable inside a Codeman-managed session + (CODEMAN_MUX=1); refuse to act otherwise. +--- + +# Driving Codeman from inside a session + +You are an agent running inside a Codeman-managed terminal session. Codeman is the +server that spawned you; its HTTP API can start, prompt, watch, and delete other +sessions. Every recipe below was verified live. Full endpoint tables and +troubleshooting: [reference/endpoints.md](reference/endpoints.md). Worked multi-worker +flows: [reference/recipes.md](reference/recipes.md). + +## 0. Guard — run this before anything else + +```bash +test "${CODEMAN_MUX:-}" = 1 || { echo "Not inside a Codeman-managed session; refusing to act."; exit 1; } +API="${CODEMAN_API_URL:?CODEMAN_API_URL not set; refusing to guess}" +SELF="${CODEMAN_SESSION_ID:?CODEMAN_SESSION_ID not set}" +# Codeman does NOT hand a session the server password. If one is set, the two +# in-reach copies are the data dir's .env (the same fallback `codeman attach` +# uses — hand-authored; nothing ever writes it) and the supervisor definition +# that install.sh wrote the password into, which is where a stock +# password-protected install actually keeps it. The data dir is wherever the +# hook-secret file lives. Values may be quoted or `export`-prefixed. +ENV_FILE="${CODEMAN_HOOK_SECRET_FILE:+${CODEMAN_HOOK_SECRET_FILE%hook-secret}.env}" +envval() { sed -n "s/^\(export \)\{0,1\}$1=//p" "$ENV_FILE" | tail -1 | sed 's/^"\(.*\)"$/\1/; s/^'\''\(.*\)'\''$/\1/'; } +if [ -z "${CODEMAN_PASSWORD:-}" ] && [ -n "$ENV_FILE" ] && [ -f "$ENV_FILE" ]; then + CODEMAN_USERNAME=$(envval CODEMAN_USERNAME) + CODEMAN_PASSWORD=$(envval CODEMAN_PASSWORD) +fi +if [ -z "${CODEMAN_PASSWORD:-}" ]; then # stock installs: install.sh puts it in the service definition + UNIT="$HOME/.config/systemd/user/codeman-web.service" + PLIST="$HOME/Library/LaunchAgents/com.codeman.web.plist" + if [ -f "$UNIT" ]; then + CODEMAN_PASSWORD=$(sed -n 's/^Environment="CODEMAN_PASSWORD=\(.*\)"$/\1/p' "$UNIT" | head -1) + elif [ -f "$PLIST" ]; then + CODEMAN_PASSWORD=$(awk '/CODEMAN_PASSWORD<\/key>/{getline; print}' "$PLIST" | sed -n 's/.*\(.*\)<\/string>.*/\1/p') + fi +fi +AUTH=(); [ -n "${CODEMAN_PASSWORD:-}" ] && AUTH=(-u "${CODEMAN_USERNAME:-admin}:$CODEMAN_PASSWORD") +CURL=(curl -sk "${AUTH[@]}") # -k: harmless on http, required on https (self-signed cert) +``` + +- If `CODEMAN_MUX` is not `1`, **stop and say so**. Do not guess an API URL; a server + you are not part of is not yours to drive. +- **A 401 is plain text, not the JSON envelope**, so on a password-protected server + every `jq` in these recipes dies with `jq: parse error` instead of showing + `UNAUTHORIZED`. If that happens, check the status with `-w '%{http_code}'`; if it + is 401 and neither fallback above found a credential, **stop and tell the user + you need credentials**. The hook-secret bypass covers only `/api/hook-event` and + `/api/status-telemetry`, never session control. +- These endpoints first ship in Codeman **1.13.0**, but do not gate on the version + number: a dev build can serve them while reporting an older version. Probe + instead: `GET .../wait` on a real session id answering 404 with an `.error` + starting `Route ` means the server predates the wait endpoints (fall back to + polling `GET .../terminal?tail=` and say so); `Session ... not found` means your + session id is wrong, not the server. + +## 1. Safety rules — read before any mutating call + +You are yourself a session on this server, and the API has **no undo**. + +- **Never act on your own session — and know that this check is the ONLY guard.** + The server has no self-protection: a session that DELETEs its own id succeeds and + dies silently (verified live). Session ids appear in both full and 8-character + forms (Docker cases export a truncated `$SELF`; mux names and UI surfaces carry + 8-char ids), so compare by prefix **in both directions**, never by equality: + + ```bash + is_self() { case "$1" in "$SELF"*) return 0 ;; esac; case "$SELF" in "$1"*) return 0 ;; esac; return 1; } + ``` + + One-directional or equality checks each miss a real combination (full `$SELF` vs + a target you transcribed in 8-char form, or truncated `$SELF` vs a full target) + and the miss deletes you. Check `is_self` before every `DELETE`, kill, respawn, + or input call. +- **Mutating calls you may make unprompted** (this is an allowlist): + `POST /api/v1/quick-start`, `POST /api/v1/sessions/:id/input`, and + `DELETE /api/v1/sessions/:id` **only** for a session you created in this + conversation, by exact id. Keep a list of the ids you create. Everything else + mutating needs the user to have asked for it. +- **Never call these** unless the user explicitly asked, naming the target: + - `DELETE /api/cases/:name` — recursively **deletes a real directory of the user's + code** from disk. One wrong case name destroys work that was never yours. + - `DELETE /api/sessions` (no id) and `DELETE /api/subagents` (no id) — bulk kills. + - respawn / ralph / orchestrator / cron mutations — respawn runs `/clear` (wipes a + conversation), orchestrator state is a single global slot, cron jobs outlive you. + - `PUT /api/settings`, `POST /api/system/update` — global UI settings; server restart. +- Never `tmux kill-session`, `pkill tmux`, `pkill claude`. The API is the only interface. +- Sessions count against a 50-session cap and case creation is uncapped: clean up every + session you start, and don't retry `quick-start` in a loop. + +## 2. Rules of the road + +- **End every input with `\r`** — literally the two characters `\r` inside the JSON + string. Codeman types the text and sends Enter **only when the input contains a + carriage return**; without it your command sits unsubmitted on the worker's prompt + and everything downstream times out. `{"input":"run the tests\r",...}`. No response + field catches this: `delivered:true` means "written to the pane", **not** + "submitted" — a `\r`-less send still reports `delivered:true` and then every wait + times out, which is why the loops below are bounded and check the terminal. +- **Single-line input only.** Newlines are stripped; one line per call. +- **Build request bodies with `jq -n` for any prompt you did not author as a + literal.** The inline `-d '{"input":"'"$P"'\r"}'` pattern breaks on the first + double quote, backslash, or `$` in a real prompt: + + ```bash + BODY=$(jq -n --arg p "$PROMPT" '{input:($p+"\r"),useMux:true,clientId:"agent-1",seq:1,wait:true,waitTimeout:60000}') + "${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" -H 'Content-Type: application/json' --data-binary "$BODY" + ``` +- **Exactly-once delivery**: always send a stable `clientId` and a monotonic + per-session `seq` on `POST .../input`. A retry after a dropped connection then + cannot double-type the prompt. Increment `seq` for each NEW input; reuse the same + pair only to re-ask about the same delivery. +- **Envelope**: success is `{"success":true,"data":…}`, errors are + `{"success":false,"error","errorCode"}`. Read `.data`. Use `/api/v1/*` paths. +- **A wait timeout is HTTP 200**, `{wait:{timedOut:true,signal:null}}` — not an error. + Loop over short waits (60 s); proxies cut long-idle connections. Timeouts are + **clamped** (ceiling 600 s): read back `wait.timeoutMs` for what was applied. +- **`stop` and `blocked` fire for `claude` sessions only** (Claude Code hooks). On + `shell`/`opencode`/`codex`/`gemini`/`antigravity`, requesting them explicitly is a + 400 — and lifecycle transitions there are coarse (a short shell command may emit + **no** `idle` transition at all, verified live), so synchronize those modes with + output markers, not signals. +- **Your typed command echoes into the output stream**, so a marker that appears + verbatim in the input line matches **before the command runs**. Always split the + marker (recipe below), keep it unique per call, and use `from=buffer` so a marker + that printed before your wait landed is still found. Matching is literal — no regex. +- **Match single space-free tokens against TUI output.** A full-screen TUI (claude, + codex, …) positions text with cursor movements, not literal spaces, so the stripped + stream can read `Yes,Itrustthisfolder` and a multi-word match is unreliable there — + whether a phrase keeps its spaces depends on how the TUI happened to draw it + (observed live: some match, some never fire). Plain command output (shell workers, + `echo` lines) keeps real spaces. + +## 3. Recipes (each verified live) + +**List sessions / find yourself** — metadata only, safe to poll: + +```bash +"${CURL[@]}" "$API/api/v1/sessions" | jq '.data[] | {id, name, mode, status}' +"${CURL[@]}" "$API/api/v1/sessions" | jq --arg s "$SELF" '.data[] | select(.id | startswith($s))' +``` + +**Start a claude worker and wait until it is actually ready.** A new session reports +`idle` before its CLI has spawned, and a brand-new case shows a **trust dialog** +first, so neither "wait for idle" nor "wait for ❯" means ready (the trust dialog +contains `❯` too — observed live). Codeman *can* auto-accept that dialog itself, but +the accept rides a stream match that misses on some runs (both outcomes seen live), +so wait for the composer first and handle the dialog only as the bounded fallback — +never send a blind Enter up front (if auto-accept already fired, it lands in the +composer). Stage 1 is short on purpose: an already-trusted case matches `bypass` in +under a second, while a **virgin case can never pass stage 1** (the dialog is up, so +the composer is not) and always pays it in full before the fallback runs — the long +budget belongs to stage 3, after the dialog is answered: + +```bash +SID=$("${CURL[@]}" -X POST "$API/api/v1/quick-start" -H 'Content-Type: application/json' \ + -d '{"caseName":"worker-1","mode":"claude"}' | jq -r '.data.sessionId') +for _ in $(seq 1 30); do # bounded: a bad SID would otherwise poll forever + [ "$("${CURL[@]}" "$API/api/v1/sessions/$SID" | jq '.data.pid')" != null ] && break; sleep 1 +done +# ⚠️ pid != null proves STARTUP only, never life: a worker that later dies inside +# its pane keeps status "idle" and a pid (the local tmux attach client, not the +# worker). The death check is wait?until=exit, below. +CID="agent-$$"; SEQ=1 +# the composer's status bar ("bypass permissions on") is the ready marker — Codeman +# spawns claude in bypass mode. Single-token matches only: TUI text is space-less. +R=$("${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \ + --data-urlencode 'match=bypass' --data-urlencode 'from=buffer' --data-urlencode 'timeout=5000') +if ! jq -e '.data.wait.matched' <<<"$R" >/dev/null; then + # composer never appeared → the trust dialog is probably still up; accept it once + T=$("${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \ + --data-urlencode 'match=trust' --data-urlencode 'from=buffer' --data-urlencode 'timeout=2000') + if jq -e '.data.wait.matched' <<<"$T" >/dev/null; then + "${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" -H 'Content-Type: application/json' \ + -d '{"input":"\r","useMux":true,"clientId":"'"$CID"'","seq":'$SEQ'}' >/dev/null + SEQ=$((SEQ+1)) + fi + R=$("${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \ + --data-urlencode 'match=bypass' --data-urlencode 'from=buffer' --data-urlencode 'timeout=45000') + jq -e '.data.wait.matched' <<<"$R" >/dev/null || \ + { echo "worker $SID never became ready; inspect terminal?tail="; } +fi +``` + +**Send a prompt and wait for the turn to finish** (claude workers — the call to +prefer). It registers the waiter *before* typing, closing the race where a separate +wait sees the previous turn's idle state. Loop by resending the **identical** request: +the repeat is a tagged duplicate (same `clientId`+`seq`) that does not retype but +answers from the session's current state. Verified: the stop hook resolves this in +seconds; a duplicate resend answers in ~20 ms without retyping. + +```bash +for TRY in $(seq 1 10); do # BOUNDED: a \r-less send never produces a signal and resends are no-op duplicates + R=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" -H 'Content-Type: application/json' \ + -d '{"input":"run the tests, then summarize in one line\r","useMux":true,"clientId":"'"$CID"'","seq":'$SEQ',"wait":true,"waitTimeout":60000}') + if jq -e '.data.wait.timedOut' <<<"$R" >/dev/null; then + [ "$TRY" = 2 ] && "${CURL[@]}" "$API/api/v1/sessions/$SID/terminal?tail=2000" \ + | jq -r '.data.terminalBuffer' | tail -5 # two straight timeouts: prompt sitting unsubmitted? + continue + fi + # Resolved — but a duplicate answering immediately reports the session's CURRENT + # state ("it is idle now"), NOT that a new turn ran. A \r-less send lands exactly + # here on try 2 (verified live), so check the terminal before believing it: + if jq -e '.data.duplicate and .data.wait.immediate' <<<"$R" >/dev/null; then + "${CURL[@]}" "$API/api/v1/sessions/$SID/terminal?tail=2000" | jq -r '.data.terminalBuffer' | tail -5 + # your prompt still on the ❯ composer line = never submitted (missing \r); + # submit it with {"input":"\r"} (the only recovery), then loop again + fi + break +done +SEQ=$((SEQ+1)); jq '.data.wait.signal, .data.status' <<<"$R" +``` + +Read the outcome in this order: `wait.signal != null` → done (`stop` is definitive; +`idle` is heuristic) — **unless** it arrived as `duplicate:true` + `immediate:true`, +which only says the session is idle *now* and must be confirmed from the terminal +(above); `wait.timedOut` → loop again (bounded); `wait.ended` → session gone, stop. +If the loop exhausts its cap, do not keep looping: read the terminal, report what +you see, and remember that a still-typed-but-unsubmitted prompt (missing `\r`) can +only be recovered by submitting it with `{"input":"\r"}`. + +**Shell worker + completion marker** — the pattern for `shell` mode (no hooks there). +The typed line must not contain the marker verbatim (the input echo would match +instantly — observed live), so build it with a variable the worker's shell expands: + +```bash +N="${RANDOM}_$$"; MARK="DONE_$N" # unique per call: tmux repaints replay old text +"${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" -H 'Content-Type: application/json' \ + -d '{"input":"M=DONE; npm run build; echo ${M}_'"$N"' rc=$?\r","useMux":true,"clientId":"'"$CID"'","seq":'$SEQ'}' +SEQ=$((SEQ+1)) +"${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \ + --data-urlencode "match=$MARK" --data-urlencode 'from=buffer' --data-urlencode 'timeout=120000' \ + | jq -r '.data.wait | {matched, snippet}' +``` + +The typed line shows `${M}_…`, the real output shows `DONE_… rc=`, and the +snippet carries the exit code back to you. + +**Read a worker's output** — the terminal buffer, tail in **bytes** (`textOutput` in +`GET .../output` stays empty for interactive sessions; don't use it): + +```bash +"${CURL[@]}" "$API/api/v1/sessions/$SID/terminal?tail=3000" | jq -r '.data.terminalBuffer' \ + | sed -e 's/\x1b\[[0-9;?]*[a-zA-Z]//g' -e 's/\x1b([B0]//g' | grep -v '^[[:space:]]*$' | tail -30 +``` + +Avoid `?full=1` (entire tmux scrollback, a context bomb) unless doing a post-mortem. + +**Detect a dead worker cheaply**: `GET .../wait?until=exit&timeout=60000` answers +immediately (`signal:"exit"`, `immediate:true`) if the PTY is gone — including a +worker that exited *inside* its pane, which `GET .../sessions/:id` keeps reporting +as `status:"idle"` with a pid (that pid is the local tmux attach client, not the +worker). The wait routes are the only liveness check; a worker dying while a wait +is parked resolves it within ~3 s. A session deleted mid-wait resolves in ~1 s. + +**Clean up** — only ids you created, `is_self`-checked, one at a time: + +```bash +is_self "$SID" || "${CURL[@]}" -X DELETE "$API/api/v1/sessions/$SID" +``` + +Everything else (endpoint tables, per-mode signal table, error codes, capacity +limits, Docker/remote caveats): [reference/endpoints.md](reference/endpoints.md). +Fan-out orchestration and blocked-worker handling: +[reference/recipes.md](reference/recipes.md). diff --git a/skills/codeman/reference/endpoints.md b/skills/codeman/reference/endpoints.md new file mode 100644 index 00000000..5d96e668 --- /dev/null +++ b/skills/codeman/reference/endpoints.md @@ -0,0 +1,214 @@ +# Codeman API reference for agents + +Loaded on demand from the `codeman` skill. Assumes the guard variables from SKILL.md +(`$API`, `$SELF`, `"${CURL[@]}"`). Canonical contract: `docs/api-reference.md` in the +Codeman repo; this file is the agent-relevant subset, verified live. + +## Envelope and errors + +Every JSON response: `{"success":true,"data":…}` or +`{"success":false,"error":"…","errorCode":"…"}`. Branch on `errorCode`: + +| `errorCode` | HTTP | Meaning | +|-------------|------|---------| +| `INVALID_INPUT` | 400 | malformed request; the message names the bad field | +| `UNAUTHORIZED` | 401 | auth required or failed (send `-u user:password`). ⚠️ The 401 body is plain text, NOT this envelope — `jq` dies with a parse error, see the guard in SKILL.md | +| `NOT_FOUND` | 404 | no such session, or one this caller does not own | +| `SESSION_BUSY` | 409 | this session's waiter cap (16, combined signal+output) is full | +| `CONFLICT` / `ALREADY_EXISTS` | 409 | conflicts with current state | +| `OPERATION_FAILED` | 422 | well-formed but could not be completed | +| `RATE_LIMITED` | 429 | per-owner or process-wide waiter pool is full — back off; switching sessions will not help | +| `INTERNAL_ERROR` | 500 | server bug | + +`SESSION_BUSY` vs `RATE_LIMITED` on the wait endpoints is deliberate: the first means +"too many waiters on *this* session", the second means the *pool* is full. + +## Sessions + +| Task | Call | +|------|------| +| list sessions (metadata only, ~1.5 KB each, safe to poll) | `GET /api/v1/sessions` | +| one session (has `.data.pid`, `null` until the PTY spawns) | `GET /api/v1/sessions/:id` — ⚠️ **not a liveness check**: a worker that dies inside its pane keeps `status:"idle"` and a pid (the tmux attach client); `wait?until=exit` is the death check | +| unified list incl. history | `GET /api/v1/sessions/unified` → `.data.sessions[]` (NOT `.data[]`), and it folds in transcript history from the whole machine — never use it to verify cleanup; `GET /api/v1/sessions` is the cleanup check | +| start case + session in one call | `POST /api/v1/quick-start` | +| send input | `POST /api/v1/sessions/:id/input` | +| read terminal (tail is in **BYTES**, raw ANSI) | `GET /api/v1/sessions/:id/terminal?tail=3000` → `.data.terminalBuffer` | +| full tmux scrollback (context bomb; post-mortems only) | `GET /api/v1/sessions/:id/terminal?full=1` | +| background agents of a session | `GET /api/v1/subagents` | +| server status / version | `GET /api/v1/status` → `.data.version` | +| delete one session (yours, `is_self`-checked) | `DELETE /api/v1/sessions/:id` | + +⚠️ `GET /api/v1/sessions/:id/output` → `.data.textOutput` looks like the obvious read +but stays **empty for interactive tmux-backed sessions** (it is fed only by the legacy +JSON-stream path). Verified empty on live claude and shell sessions. Read +`terminal?tail=` instead and strip ANSI: + +```bash +… | jq -r '.data.terminalBuffer' | sed -e 's/\x1b\[[0-9;?]*[a-zA-Z]//g' -e 's/\x1b([B0]//g' +``` + +`POST /api/v1/quick-start` body (all optional): +`{"caseName":"worker-1","mode":"claude","sessionName":"w9-worker","effort":"high"}` +— `mode` ∈ `claude|shell|opencode|codex|gemini|antigravity`; response is +`.data.{sessionId, caseName, casePath}`. Creates the case directory (a real directory +on the user's disk) if missing — do not retry it in a loop, and remember the name. + +`POST /api/v1/sessions/:id/input` body: +`{"input":"one line\r","useMux":true,"clientId":"agent-1","seq":1}` plus optionally +`"wait"` / `"waitTimeout"` (below). + +- ⚠️ **The input must contain `\r`** (the JSON escape, i.e. a real carriage return) + **or Enter is never sent**: the text is typed onto the worker's prompt and sits + there unsubmitted. Verified live — this is the number-one silent failure, and no + response field catches it: `delivered:true` means "written to the pane", not + "submitted". A `\r`-less send with `wait` reports `delivered:true` and then every + wait on that turn times out. Without `wait`, fire-and-forget returns an **empty** + `{"success":true,"data":{}}` — no `delivered`, no `duplicate`; those fields exist + only on the `wait` variant, so a fire-and-forget flow gets no delivery + confirmation at all. +- `input` must be single-line (newlines are stripped). To send a bare Enter (confirm + a dialog), send `{"input":"\r"}`. +- `clientId`+`seq` give exactly-once delivery: the server applies each pair at most + once. Increment `seq` per new input. + +## The wait primitives + +Three bounded long-polls. Shared semantics: + +- **Timeout = HTTP 200** with `wait.timedOut:true`. Loop over short waits (60 s); + `tailscale serve` / cloudflared cut idle connections. +- Timeouts are **clamped** to `[1000, 600000]` ms (operator-tunable); the applied + value is echoed as `wait.timeoutMs` — read it back, never assume. +- All three nest the result under `.data.wait`, same shape, so one helper parses all. +- `.data.status` (post-wait `SessionStatus`) and `.data.limitPaused` ride along. + `limitPaused:true` means the session is paused on a usage limit and will emit + nothing until reset — a timeout is then *expected*; do not retry hard, and do not + kill the worker. + +### Signals by mode + +| Signal | Meaning | Available for | +|--------|---------|---------------| +| `idle` | output stabilized + prompt detected — heuristic, can flap mid-turn | every mode | +| `working` | session started producing output | every mode | +| `stop` | Claude Code `stop` hook — the definitive end-of-turn | `claude` only | +| `blocked` | `permission_prompt` / `elicitation_dialog` hook — the worker needs an answer | `claude` only | +| `exit` | PTY exited or session deleted | every mode | + +Default `until` set: `stop,idle,exit`. On non-claude modes the server silently drops +`stop`/`blocked` from the *default* set (echoed back as `wait.until`, e.g. +`["idle","exit"]` on shell); requesting them *explicitly* there is a 400 naming the +mode. ⚠️ On hook-less modes the lifecycle signals are also **coarse in practice**: a +short shell command produced **no** `idle` transition within 60 s (verified live), so +a `fresh=1` / fresh-delivery wait can burn its whole timeout while the work finished +long ago. Synchronize hook-less modes with `wait-output` markers instead. + +Two more places hooks go missing even in claude mode: **Docker cases** need +`CODEMAN_DOCKER_BRIDGE_HOOKS=1` on the server (without it only `idle`/`working`/ +`exit` arrive), and **remote-SSH cases** run the agent on another host whose hooks may +never reach this server. When unsure, ask for `stop,idle,exit`. + +⚠️ **Signals are edge-triggered with no history.** A signal that fires while no +waiter is registered is gone; no later wait can observe it (`until=stop` on a worker +whose turn already ended just times out, with or without `fresh` — verified live). +Register the waiter before the event can happen: send-and-wait does exactly that, +and `wait-output` markers with `from=buffer` are latched by construction. Never +fire-and-forget N prompts and then gather signal-waits worker by worker; every +worker that finishes before its gather is unobservable (see recipes.md Flow 3b). + +### `GET /api/v1/sessions/:id/wait` + +| Param | Default | Notes | +|-------|---------|-------| +| `until` | `stop,idle,exit` | comma list; unknown token → 400 naming it | +| `timeout` | 60000 | ms, clamped; applied value echoed as `wait.timeoutMs` | +| `fresh` | `0` | `1` requires an actual *transition*, ignoring the state at call time | + +⚠️ A session whose PTY has not spawned (`pid:null`) or has exited counts as `exit` +**right now**: with the default set the call answers immediately +(`signal:"exit", immediate:true`). That is how you detect a dead worker cheaply — but +it also means "wait for my just-created session" needs the readiness recipe in +SKILL.md, not this endpoint. + +### `GET /api/v1/sessions/:id/wait-output` + +| Param | Default | Notes | +|-------|---------|-------| +| `match` | required | literal substring, 1–200 chars, ANSI-stripped; chunk-straddling matches found; **no regex** — a `regex=` param is a 400 | +| `nocase` | `0` | case-insensitive compare; snippet keeps original casing | +| `from` | `now` | `buffer` scans the tail (~256 KB) of existing output first | +| `timeout` | 60000 | same clamp | + +Four traps, all observed live: + +1. **The echo of your own typed command is output.** A marker appearing verbatim in + the input line matches the moment the text is typed, before the command runs. + Split the marker with a shell variable: send `M=DONE; …; echo ${M}_1234\r`, wait + on `DONE_1234`. +2. **`from=now` misses text printed before the wait landed** — a marker echoed just + before the request registered timed out at full length. After sending a command, + always wait with `from=buffer`. +3. **`from=now` can also match too much**: tmux repaints old screen content as + ordinary output on attach/resize/redraw, so a *generic* marker (`BUILD OK`) + matches stale text. Unique-per-call markers (`DONE_$RANDOM`) make both `from` + modes safe. +4. **TUI output can be space-less in the stream.** Full-screen TUIs (claude, codex, + …) position words with cursor-movement escapes rather than literal spaces, so + the stripped stream can read `Yes,Itrustthisfolder` while the pane shows the + spaced phrase. Whether a given phrase keeps its spaces depends on how the TUI + drew it (observed live: some multi-word matches fire, some never do), so treat + multi-word matches against TUI screens as unreliable and match a **single + space-free token** (`trust`, `bypass`). Plain command output (shell workers, + `echo` lines) keeps real spaces and multi-word matches work there. + +Build the query with `-G --data-urlencode` (a `+` in a hand-built query decodes to a +space). Result extras: `wait.matched`, `wait.match`, `wait.snippet` (bounded window +around the match, blank runs collapsed — the snippet is often all you need to read). + +### `POST /api/v1/sessions/:id/input` with `wait` + +| Field | Notes | +|-------|-------| +| `wait` | `true` (default signal set) or the same comma grammar as `until`; absent = historical fire-and-forget | +| `waitTimeout` | ms, same clamp | + +Registers the waiter **before** typing, which closes the race where send-then-wait +sees the previous turn's idle state and returns instantly. Response adds `delivered` +and `duplicate` beside the standard `wait` object. + +A **tagged duplicate** (same `clientId`+`seq` already applied) does not retype but +still honors `wait`, answering from the session's *current* state instead of +requiring a new transition (`delivered:false, duplicate:true` — verified: ~20 ms, +command ran exactly once). That is what makes the resend-identical-request loop in +SKILL.md correct: iteration 1 delivers and needs a transition; later iterations +resolve immediately if the turn ended in between. ⚠️ The flip side: a duplicate's +`immediate:true` answer is the current state and nothing more — an idle worker +whose prompt was never submitted (missing `\r`) produces the same +`signal:"idle", immediate:true` as one that finished the turn. Confirm from +`terminal?tail=` before reporting success; SKILL.md's loop shows where. + +### Outcome parsing, in order + +1. `wait.signal != null` (or `wait.matched == true`) — the thing happened. + `wait.immediate:true` rides along and means the condition already held at call + time; if that is not what you meant, you wanted `fresh=1` or send-and-wait. +2. `wait.timedOut` — poll boundary; loop again. +3. `wait.ended` — session deleted/torn down mid-wait; stop looping. + +## Troubleshooting + +| Symptom | Cause / fix | +|---------|-------------| +| every curl fails with a certificate error | you dropped `-k`; `CODEMAN_API_URL` is HTTPS with a self-signed cert | +| `jq: parse error` on every call | plain-text 401s: the server has a password. Check with `-w '%{http_code}'`, use the guard's `.env` fallback, and if no `.env` exists, stop and ask the user for credentials | +| input arrives but nothing happens; later waits all time out | the input had no `\r`, so Enter was never sent; the text is sitting on the worker's prompt. **Submitting it with `{"input":"\r"}` is the ONLY recovery** — Ctrl+U (0x15) and Esc do NOT clear the composer (verified live) — and the flush costs one turn in which the worker reasons about the junk; open the next real prompt with "ignore the garbled line above:" | +| `GET .../sessions/$CODEMAN_SESSION_ID` 404s | Docker case: the env id is truncated to 8 chars; find yourself with `startswith($SELF)`, and always self-compare by prefix, in both directions | +| `CODEMAN_MUX` unset but you seem to be in a session | remote-SSH case: the env vars are not exported there. Fail closed — refuse to act | +| connection refused from inside a container | a loopback-bound server is unreachable from a container, and `CODEMAN_DOCKER_BRIDGE_HOOKS=1` does **not** fix that: it opens a hooks-only listener, so hook events start flowing but `/api/v1/*` stays refused. Driving the API from inside a Docker case needs a reachable bind (an operator decision); report it, don't retry | +| wait routes 404 on a valid session id | read the `.error` text: a `Route ...` prefix means the server predates the wait endpoints (< 1.13.0; a dev build can serve them while reporting an older version, so probe, never version-compare) — poll `terminal?tail=` and say so. `Session ... not found` means your id is wrong, not the server | +| wait on `stop` never resolves | non-claude mode, or hooks not reaching the server (Docker/remote), or a case created by Codeman < 1.13.0 against an `--https` install (its hook curls lacked `-k` and TLS-failed silently; a 1.13.0+ server rewrites them the next time a session starts in that case). Use markers or `idle,exit` | +| new claude worker ignores its first prompt | it was showing the first-run trust dialog and Codeman's auto-accept missed; use the readiness recipe in SKILL.md (wait for `bypass` first, accept the dialog only as the bounded fallback) | +| `wait-output` times out although the pane shows the text | multi-word match against a TUI screen; the stream has no spaces there — match one token | +| `wait-output` matched instantly with stale text | generic marker + tmux repaint; use `DONE_$RANDOM` | +| 409 `SESSION_BUSY` on a wait | too many concurrent waiters on that session (cap 16 combined); reuse one wait per worker | +| 429 `RATE_LIMITED` on a wait | global/owner waiter pool full; back off, do not switch sessions | diff --git a/skills/codeman/reference/recipes.md b/skills/codeman/reference/recipes.md new file mode 100644 index 00000000..4bbb2fa5 --- /dev/null +++ b/skills/codeman/reference/recipes.md @@ -0,0 +1,249 @@ +# Worked orchestration flows + +Loaded on demand from the `codeman` skill. Every flow assumes the guard preamble from +SKILL.md ran (`$API`, `$SELF`, `"${CURL[@]}"`, `is_self`). Track every session id you +create; delete them (and only them) when done. Remember the two silent killers: +**every input ends with `\r`**, and **markers must be split** so the typed-line echo +does not match them. + +## Flow 1: claude worker, end to end + +Start a worker, get it truly ready (trust dialog included), give it a task, wait for +the turn to finish, read the answer, clean up. Verified live: the stop hook resolves +the send-and-wait within seconds of the turn ending. + +```bash +# 1. start (returns before the CLI inside is ready) +SID=$("${CURL[@]}" -X POST "$API/api/v1/quick-start" -H 'Content-Type: application/json' \ + -d '{"caseName":"worker-tests","mode":"claude"}' | jq -r '.data.sessionId') +CREATED+=("$SID") # the cleanup list +CID="agent-$$"; SEQ=1 + +# 2. readiness. "wait for idle" or "wait for ❯" is NOT readiness: a fresh session +# reports idle before anything spawned, and the first-run trust dialog contains ❯. +# Codeman CAN auto-accept that dialog, but the accept misses on some runs (both +# outcomes seen live), so: composer marker first, dialog only as the bounded +# fallback (a blind Enter up front would land in an already-ready composer). +# Stage 1 is SHORT on purpose: an already-trusted case matches in <1 s, while a +# virgin case can never pass it (the dialog is up) and always pays it in full — +# the long budget belongs to stage 3, after the dialog is answered. +# Single-token matches only: TUI text is space-less in the stream. +for _ in $(seq 1 30); do + [ "$("${CURL[@]}" "$API/api/v1/sessions/$SID" | jq '.data.pid')" != null ] && break; sleep 1 +done +# (pid != null proves startup only — a worker that later dies inside its pane keeps +# status "idle" and a pid. The death check is wait?until=exit.) +R=$("${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \ + --data-urlencode 'match=bypass' --data-urlencode 'from=buffer' --data-urlencode 'timeout=5000') +if ! jq -e '.data.wait.matched' <<<"$R" >/dev/null; then + T=$("${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \ + --data-urlencode 'match=trust' --data-urlencode 'from=buffer' --data-urlencode 'timeout=2000') + if jq -e '.data.wait.matched' <<<"$T" >/dev/null; then + "${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" -H 'Content-Type: application/json' \ + -d '{"input":"\r","useMux":true,"clientId":"'"$CID"'","seq":'$SEQ'}' >/dev/null + SEQ=$((SEQ+1)) + fi + R=$("${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \ + --data-urlencode 'match=bypass' --data-urlencode 'from=buffer' --data-urlencode 'timeout=45000') + jq -e '.data.wait.matched' <<<"$R" >/dev/null || echo "worker $SID not ready; inspect terminal?tail=" +fi + +# 3. send-and-wait, looping on the IDENTICAL request (tagged duplicate: no retype). +# BOUNDED (a \r-less send would otherwise loop forever), body built with jq -n so +# quotes/backslashes/$ in a real prompt survive; note the appended \r. +PROMPT='run the unit tests and summarize failures in one line' +BODY=$(jq -n --arg p "$PROMPT" --arg c "$CID" --argjson s "$SEQ" \ + '{input:($p+"\r"),useMux:true,clientId:$c,seq:$s,wait:true,waitTimeout:60000}') +for TRY in $(seq 1 10); do + R=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" \ + -H 'Content-Type: application/json' --data-binary "$BODY") + if jq -e '.data.wait.timedOut' <<<"$R" >/dev/null; then + jq -e '.data.limitPaused' <<<"$R" >/dev/null && sleep 60 # usage-limit pause: silence is expected + [ "$TRY" = 2 ] && "${CURL[@]}" "$API/api/v1/sessions/$SID/terminal?tail=2000" \ + | jq -r '.data.terminalBuffer' | tail -5 # is the prompt sitting unsubmitted? + continue + fi + # Resolved — but duplicate + immediate is only "the session is idle NOW", which a + # never-submitted (\r-less) prompt also produces. Check before believing it: + if jq -e '.data.duplicate and .data.wait.immediate' <<<"$R" >/dev/null; then + "${CURL[@]}" "$API/api/v1/sessions/$SID/terminal?tail=2000" \ + | jq -r '.data.terminalBuffer' | tail -5 + # prompt still on the ❯ composer line = never submitted; {"input":"\r"} is the + # only recovery, then loop again + fi + break +done +SEQ=$((SEQ+1)) + +# 4. interpret +case "$(jq -r '.data.wait.signal' <<<"$R")" in + stop) : ;; # definitive end of turn + idle) : ;; # heuristic — and if it rode a duplicate with + # immediate:true, it proves nothing ran (step 3) + exit) echo "worker died" ;; + null) jq -e '.data.wait.ended' <<<"$R" >/dev/null && echo "worker deleted mid-wait" ;; +esac + +# 5. read the answer: terminal tail (BYTES), ANSI-stripped. textOutput stays empty +# for interactive sessions; terminal?full=1 is a context bomb. +"${CURL[@]}" "$API/api/v1/sessions/$SID/terminal?tail=4000" | jq -r '.data.terminalBuffer' \ + | sed -e 's/\x1b\[[0-9;?]*[a-zA-Z]//g' -e 's/\x1b([B0]//g' | grep -v '^[[:space:]]*$' | tail -30 + +# 6. clean up — exact id, own list only, self-check +is_self "$SID" || "${CURL[@]}" -X DELETE "$API/api/v1/sessions/$SID" +``` + +Increment `SEQ` for every *new* input to the same worker. Reuse the same `SEQ` only to +re-ask about the same delivery (the duplicate-wait loop above). + +## Flow 2: shell worker running a build, marker-synchronized + +`shell` sessions have no hooks (`stop`/`blocked` are a 400 there), and their lifecycle +signals are coarse — a short command may emit no `idle` transition at all (verified +live), so send-and-wait can burn its whole timeout. The reliable pattern is a split, +unique marker plus `wait-output from=buffer`: + +```bash +SID=$("${CURL[@]}" -X POST "$API/api/v1/quick-start" -H 'Content-Type: application/json' \ + -d '{"caseName":"builder","mode":"shell"}' | jq -r '.data.sessionId') +CREATED+=("$SID") +for _ in $(seq 1 30); do + [ "$("${CURL[@]}" "$API/api/v1/sessions/$SID" | jq '.data.pid')" != null ] && break; sleep 1 +done + +# Split marker: the typed line carries ${M}_N, only the OUTPUT carries DONE_N. +# An unsplit marker matches the echo of your own keystrokes before the build runs. +N="${RANDOM}_$$"; MARK="DONE_$N" +"${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" -H 'Content-Type: application/json' \ + -d '{"input":"M=DONE; npm run build; echo ${M}_'"$N"' rc=$?\r","useMux":true,"clientId":"build-'$$'","seq":1}' + +for TRY in $(seq 1 30); do # BOUNDED (30 min): a \r-less send makes an uncapped loop infinite + R=$("${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \ + --data-urlencode "match=$MARK" --data-urlencode 'from=buffer' --data-urlencode 'timeout=60000') + jq -e '.data.wait.matched' <<<"$R" >/dev/null && break + jq -e '.data.wait.ended' <<<"$R" >/dev/null && { echo "worker gone"; break; } + [ "$TRY" = 2 ] && "${CURL[@]}" "$API/api/v1/sessions/$SID/terminal?tail=2000" \ + | jq -r '.data.terminalBuffer' | tail -5 # command still sitting unsubmitted? +done +jq -r '.data.wait.snippet' <<<"$R" # e.g. "DONE_123_456 rc=0" — the exit code rides the marker line +``` + +## Flow 3: fan out N workers, gather as each finishes + +Start everything first, then gather. One in-flight wait per worker — the per-session +waiter cap is 16 and abandoned concurrent waits pile up against it. + +```bash +declare -A WORKER MARKS +for task in lint typecheck unit; do + SID=$("${CURL[@]}" -X POST "$API/api/v1/quick-start" -H 'Content-Type: application/json' \ + -d '{"caseName":"fan-'"$task"'","mode":"shell"}' | jq -r '.data.sessionId') + WORKER[$task]=$SID; CREATED+=("$SID") +done +for task in "${!WORKER[@]}"; do + SID=${WORKER[$task]} + for _ in $(seq 1 30); do + [ "$("${CURL[@]}" "$API/api/v1/sessions/$SID" | jq '.data.pid')" != null ] && break; sleep 1 + done + N="${task}_${RANDOM}"; MARKS[$task]="DONE_$N" + "${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" -H 'Content-Type: application/json' \ + -d '{"input":"M=DONE; npm run '"$task"'; echo ${M}_'"$N"' rc=$?\r","useMux":true,"clientId":"fan-'$$'","seq":1}' +done +for task in "${!WORKER[@]}"; do # sequential gather; each wait blocks until that worker is done + for TRY in $(seq 1 30); do # BOUNDED per worker, same reasoning as Flow 2 + R=$("${CURL[@]}" -G "$API/api/v1/sessions/${WORKER[$task]}/wait-output" \ + --data-urlencode "match=${MARKS[$task]}" --data-urlencode 'from=buffer' --data-urlencode 'timeout=60000') + jq -e '.data.wait.matched or .data.wait.ended' <<<"$R" >/dev/null && break + done + echo "$task: $(jq -r '.data.wait.snippet // "worker gone"' <<<"$R" | tail -1)" +done +``` + +## Flow 3b: fan out N CLAUDE workers + +Send-and-wait is synchronous, so the shell-flow shape ("send everything, then +gather") does not translate directly: the send *is* the wait, and worker 2's prompt +would not go out until worker 1's turn ended. Two working patterns, both verified +live (and one anti-pattern, measured failing, replaced by B): + +**A. Background the send-and-waits** (simplest; each resolved on `stop` while the +other was still running): + +```bash +sendwait() { # $1=sid $2=prompt $3=seq — assumes the worker passed Flow 1's readiness + local body; body=$(jq -n --arg p "$2" --argjson s "$3" \ + '{input:($p+"\r"),useMux:true,clientId:"fan-'$$'",seq:$s,wait:true,waitTimeout:600000}') + "${CURL[@]}" -X POST "$API/api/v1/sessions/$1/input" \ + -H 'Content-Type: application/json' --data-binary "$body" > "/tmp/fan-$1.json" +} +( sendwait "$SID1" 'refactor module A and reply DONE' 2 & \ + sendwait "$SID2" 'write tests for module B and reply DONE' 2 & wait ) +jq -c '.data.wait | {signal, waitedMs}' /tmp/fan-"$SID1".json /tmp/fan-"$SID2".json +``` + +One in-flight wait per worker keeps you far from the 16-per-session waiter cap. + +**B. Fire-and-forget, then gather with output markers.** If you must send every +prompt before waiting on anything, do **not** gather with signal waits: signals +are edge-triggered with no history, so a `stop` that fires before the gather +reaches that worker is gone and unobservable afterwards — `fresh=1` cannot help, +and neither can omitting it (measured: worker 2's turn ended at +2 s, its +sequential `until=stop,exit&fresh=1` gather burned its full bounded 300 s and +reported nothing). Gather instead on a marker each worker prints itself, which +`from=buffer` re-finds no matter when it appeared: + +```bash +# SIDS[1], SIDS[2] = worker ids that already passed Flow 1's readiness. +# The typed prompt must NOT contain the finished marker verbatim (your keystrokes +# echo into the output stream and would match instantly), so ask for it in halves: +declare -A TOK +for i in 1 2; do + TOK[$i]="${RANDOM}_$i" + BODY=$(jq -n --arg p "do task $i; when completely done print the word WORKDONE immediately followed by _${TOK[$i]}" \ + --arg c "fan-$$" --argjson s 2 '{input:($p+"\r"),useMux:true,clientId:$c,seq:$s}') + "${CURL[@]}" -X POST "$API/api/v1/sessions/${SIDS[$i]}/input" \ + -H 'Content-Type: application/json' --data-binary "$BODY" +done +for i in 1 2; do # order no longer matters: the marker is latched in the buffer + "${CURL[@]}" -G "$API/api/v1/sessions/${SIDS[$i]}/wait-output" \ + --data-urlencode "match=WORKDONE_${TOK[$i]}" --data-urlencode 'from=buffer' \ + --data-urlencode 'timeout=600000' | jq -c '.data.wait | {matched, snippet}' +done +``` + +Use A unless you genuinely need to send everything before waiting on anything: A +needs no marker discipline, and resolves on the definitive `stop` instead of on +the worker remembering to print a token. + +## Flow 4: watch for a worker stuck on a permission prompt + +Claude workers can block on a permission dialog. `blocked` is a wait signal +(claude-mode only), so watch for it and surface the question to the user instead of +guessing an answer: + +```bash +R=$("${CURL[@]}" "$API/api/v1/sessions/$SID/wait?until=stop,blocked,exit&timeout=60000") +if [ "$(jq -r '.data.wait.signal' <<<"$R")" = blocked ]; then + "${CURL[@]}" "$API/api/v1/sessions/$SID/terminal?tail=2000" | jq -r '.data.terminalBuffer' \ + | sed -e 's/\x1b\[[0-9;?]*[a-zA-Z]//g' | grep -v '^[[:space:]]*$' | tail -15 + # show this to the user and ask how to answer; do NOT auto-confirm another + # session's permission prompt +fi +``` + +## Cleanup discipline + +At the end of the conversation (or on abort), delete exactly what you created: + +```bash +for id in "${CREATED[@]}"; do + is_self "$id" || "${CURL[@]}" -X DELETE "$API/api/v1/sessions/$id" +done +``` + +- Only ids from your own `CREATED` list. Never enumerate `/api/v1/sessions` and + delete by pattern; other sessions belong to the user. +- If you created a *case* purely as scratch and the user confirmed it is disposable, + `DELETE /api/v1/cases/:name` removes it — but that recursively deletes the + directory from disk, so never do it without the user's explicit go-ahead for that + exact name. diff --git a/src/config/agent-wait.ts b/src/config/agent-wait.ts new file mode 100644 index 00000000..6cf8530a --- /dev/null +++ b/src/config/agent-wait.ts @@ -0,0 +1,112 @@ +/** + * @fileoverview Bounds for the agent wait primitives. + * + * These back the blocking endpoints an agent uses to orchestrate other sessions + * (`GET /api/sessions/:id/wait`, `GET /api/sessions/:id/wait-output`, and the + * `wait` field on `POST /api/sessions/:id/input`). Plan: `docs/agent-control-plan.md`. + * + * Why every value is bounded: + * - An unbounded long-poll is a socket leak. A caller that asks for a 12-hour wait + * and walks away holds a connection (and a waiter, and a timer) until the process + * restarts, so `MAX_WAIT_MS` is a hard ceiling applied server-side. + * - `DEFAULT_WAIT_MS` is deliberately short (60s). Production is reached through + * `tailscale serve` and users also run cloudflared tunnels; both can cut an idle + * connection, so the documented pattern is a client-side loop over short waits + * rather than one very long call. Fastify itself is happy to hold the request + * (`requestTimeout` defaults to 0, and `keepAliveTimeout` applies between + * requests, not to an in-flight one), the intermediaries are the constraint. + * - The waiter caps mirror `MAX_SSE_CLIENTS` in `map-limits.ts`: each pending + * waiter costs an open HTTP response plus a timer, so the pool is capped rather + * than queued. Exceeding a cap is an explicit error, never a silent wait. + * - There are THREE caps, not two, because a process-wide pool with no per-user + * dimension lets one user deny the primitive to everyone else. `middleware/auth.ts` + * already treats that shape as a bug (its `userFailures` bucket exists so "one user + * behind a NAT can't lock out everyone else"); `MAX_WAITERS_PER_OWNER` is the same + * idea for waiters. It applies only when the caller has an owner, so single-user + * mode is byte-identical to having no owner cap at all. + * + * All values are env-overridable and clamped to sane hard bounds, so a typo in an + * env var degrades to the default instead of disabling the protection. + * + * @module config/agent-wait + */ + +/** Absolute floor for any wait, in ms. Sub-second waits are polling, not waiting. */ +export const MIN_WAIT_MS = 1_000; + +/** Ceiling the operator-configurable maximum is itself clamped to. */ +const HARD_MAX_WAIT_MS = 3_600_000; + +function envInt(name: string, fallback: number, min: number, max: number): number { + const raw = parseInt(process.env[name] || '', 10); + if (!Number.isFinite(raw) || raw <= 0) return fallback; + return Math.max(min, Math.min(max, raw)); +} + +/** Longest a single wait may block. Requests above this are clamped down, not rejected. */ +export const MAX_WAIT_MS = envInt('CODEMAN_WAIT_MAX_MS', 600_000, MIN_WAIT_MS, HARD_MAX_WAIT_MS); + +/** Used when the caller omits `timeout`. Never exceeds MAX_WAIT_MS. */ +export const DEFAULT_WAIT_MS = Math.min( + envInt('CODEMAN_WAIT_DEFAULT_MS', 60_000, MIN_WAIT_MS, HARD_MAX_WAIT_MS), + MAX_WAIT_MS +); + +/** Concurrent waiters (signal + output) allowed against one session. */ +export const MAX_WAITERS_PER_SESSION = envInt('CODEMAN_WAIT_MAX_PER_SESSION', 16, 1, 256); + +/** + * Ceiling the operator-configurable total is itself clamped to. + * + * 512 rather than the 4096 this started at. Every other knob in this file degrades + * safely on a bad value; a 4096 ceiling instead lets a well-meaning operator turn the + * protection into the problem, since 4096 concurrent held responses (each an open + * socket, a timer and a pending promise) exceeds the 1024 soft `RLIMIT_NOFILE` that is + * still the default on most Linux distros, before counting PTYs, SSE clients and + * WebSockets. 512 is ~5x `MAX_SSE_CLIENTS` (100, the pool this one is modelled on), so + * the knob stays useful for a busy orchestration host while the whole server still fits + * inside a default fd budget with room to spare. + */ +const HARD_MAX_WAITERS_TOTAL = 512; + +/** Concurrent waiters allowed across every session in the process. */ +export const MAX_WAITERS_TOTAL = envInt('CODEMAN_WAIT_MAX_TOTAL', 128, 1, HARD_MAX_WAITERS_TOTAL); + +/** + * Concurrent waiters allowed for one owner (multi-user mode's `Session.owner`). + * + * Sits between the per-session cap (16) and the process-wide one (128): high enough + * that one user orchestrating several workers at once never trips it, low enough that + * a single user cannot occupy the whole pool and deny the primitive to everyone else, + * admin included. Ignored entirely when the caller has no owner, which is every + * request in single-user mode. + */ +export const MAX_WAITERS_PER_OWNER = envInt('CODEMAN_WAIT_MAX_PER_OWNER', 48, 1, HARD_MAX_WAITERS_TOTAL); + +/** Bounds on the literal `match` string accepted by wait-output. */ +export const MIN_MATCH_LENGTH = 1; +export const MAX_MATCH_LENGTH = 200; + +/** + * Tail of the terminal buffer scanned by `wait-output?from=buffer`. + * + * The buffer itself runs to 32MB. Scanning all of it would be an ANSI strip over + * 32MB (a full second copy) on a request an agent may issue in a loop, and the + * question `from=buffer` answers is "did this appear recently", not "ever". The + * tail is continuous with the live stream, since `_terminalBuffer.append(data)` + * and `emit('terminal', data)` receive the same bytes. + */ +export const MAX_BUFFER_SCAN_BYTES = envInt('CODEMAN_WAIT_BUFFER_SCAN_BYTES', 256 * 1024, 4 * 1024, 8 * 1024 * 1024); + +/** Characters of surrounding output returned either side of a wait-output match. */ +export const MAX_SNIPPET_CONTEXT = 80; + +/** + * Clamp a caller-supplied timeout into [MIN_WAIT_MS, MAX_WAIT_MS]. + * Absent / non-numeric / non-finite input falls back to DEFAULT_WAIT_MS. + */ +export function clampWaitMs(value: unknown): number { + const n = typeof value === 'string' ? Number(value) : value; + if (typeof n !== 'number' || !Number.isFinite(n)) return DEFAULT_WAIT_MS; + return Math.max(MIN_WAIT_MS, Math.min(MAX_WAIT_MS, Math.trunc(n))); +} diff --git a/src/hooks-config.ts b/src/hooks-config.ts index 9c33bf6a..f78279cb 100644 --- a/src/hooks-config.ts +++ b/src/hooks-config.ts @@ -173,7 +173,11 @@ export function generateHooksConfig(): { hooks: Record } { const curlCmd = (event: HookEventType) => `HOOK_DATA=$(cat 2>/dev/null || echo '{}'); ` + `printf '{"event":"${event}","sessionId":"%s","data":%s}' "$CODEMAN_SESSION_ID" "$HOOK_DATA" | ` + - `curl -s -X POST "$CODEMAN_API_URL/api/hook-event" ` + + // `-k`, same as the statusline exporter: CODEMAN_API_URL is loopback HTTPS with + // a self-signed cert on --https/tailscale installs. Without it curl exits 60, + // the `|| true` swallows it, and ALL SIX hook events die silently: respawn loses + // its definitive idle signals and the wait endpoints lose stop/blocked. + `curl -sk -X POST "$CODEMAN_API_URL/api/hook-event" ` + `-H 'Content-Type: application/json' ` + `-H "X-Codeman-Hook-Secret: $(cat "$CODEMAN_HOOK_SECRET_FILE" 2>/dev/null)" ` + `--data @- ` + @@ -433,8 +437,10 @@ export async function writeHooksConfig(casePath: string): Promise { * X-Codeman-Hook-Secret header was added (COD-54, 2026-06-10) keep hook curls in their * settings.local.json that POST to /api/hook-event WITHOUT the secret — which, once the * gate requires it unconditionally (COD-91), silently 401 on a password-protected install. - * Older Codeman blocks also lack the background Bash async-rewake hook. Refresh either - * stale shape on launch so existing cases gain both current behaviors. + * Older Codeman blocks also lack the background Bash async-rewake hook. A third stale + * shape: hook curls without `-k`, which exit 60 on every --https/tailscale install (the + * cert is self-signed), swallowed by the hooks' own `|| true` — all six hook events die + * silently. Refresh any of these stale shapes on launch so existing cases heal. * * Deliberately surgical: regenerates ONLY when settings.local.json already contains * Codeman's own hook curls (they target `/api/hook-event`) and they are stale. No-op @@ -457,7 +463,11 @@ export async function refreshStaleCodemanHooks(casePath: string): Promise // absence on our own hooks means they predate COD-54 and need regenerating. const hasSecret = hooksJson.includes('X-Codeman-Hook-Secret'); const hasBackgroundWake = hooksJson.includes(BACKGROUND_WAKE_MARKER); - if (!isOurs || (hasSecret && hasBackgroundWake)) return; + // The pre--k curl shape: `curl -sk -X POST` does not contain `curl -s -X POST` + // as a substring, so this cleanly identifies hook curls that die with exit 60 + // on a self-signed HTTPS install. + const hasTlsFlaglessCurl = hooksJson.includes('curl -s -X POST'); + if (!isOurs || (hasSecret && hasBackgroundWake && !hasTlsFlaglessCurl)) return; const generated = generateHooksConfig(); const merged = { ...existing, diff --git a/src/session-cli-builder.ts b/src/session-cli-builder.ts index 1e970c45..2461c841 100644 --- a/src/session-cli-builder.ts +++ b/src/session-cli-builder.ts @@ -121,7 +121,10 @@ export function buildClaudeEnv(sessionId: string): Record(schema: z.ZodType, query: unknown, label: string): T { + const result = schema.safeParse(query); + if (result.success) return result.data; + const issue = result.error.issues[0]; + const field = issue && issue.path.length > 0 ? issue.path.join('.') : ''; + const detail = issue?.message ?? 'validation failed'; + const message = field ? `Invalid ${label} parameter '${field}': ${detail}` : `Invalid ${label} parameters: ${detail}`; + throw Object.assign(new Error(message), { + statusCode: 400, + body: createErrorResponse(ApiErrorCode.INVALID_INPUT, message), + }); +} + +/** + * Map a waiter-cap rejection to the code that tells the caller the truth. + * + * The two caps mean different things and warrant different recovery: `session` is + * genuinely about THIS session, while `owner` and `total` are process-wide budgets + * that say nothing about it. Reporting a global cap as `SESSION_BUSY` (409, + * documented as "Session is busy") sent an agent off to a different session to hit + * the identical error. `RATE_LIMITED` is the code whose whole meaning is "come back + * later", and clients and proxies already treat 429 that way. + */ +function waitCapacityResponse(err: WaitCapacityError): ApiResponse { + const code = err.scope === 'session' ? ApiErrorCode.SESSION_BUSY : ApiErrorCode.RATE_LIMITED; + // The registry's message already names the scope and the limit; passing it through + // verbatim keeps the wording in one place. + return createErrorResponse(code, err.message); +} + +/** + * The signal a session is ALREADY emitting, corrected for liveness. + * + * `signalForStatus` alone is not enough here, because `Session` parks a DEAD PTY at + * `_status = 'idle'` (both `onExit` handlers do) and the object survives in the + * session map until an explicit DELETE. Trusting the status therefore answers the + * default wait with `{signal:"idle", immediate:true}` for a worker that has + * crashed — HTTP 200, no error anywhere, and the agent types its next prompt into a + * corpse — while `until=exit` blocks for the full timeout on an event that already + * happened and can never happen again. + * + * `pid === null` means no process is behind this session: it exited, it was + * detached, or it was created and never started. All three are `exit` from a + * caller's point of view — nothing is running — and in all three the agent's + * correct next move is to (re)start the worker rather than to type at it. The + * response still carries the raw `status` alongside, so nothing is hidden. + * + * ⚠️ `pid` alone is NOT enough, and on the normal configuration it is never the + * thing that fires — see `workerIsDead()`. `dead` carries the mux layer's answer. + * + * Fixing it HERE rather than in `signalForStatus` is deliberate: liveness is not + * derivable from `SessionStatus`, and the registry holds no `Session` reference. + */ +function currentSignalFor(session: { pid: number | null; status: SessionStatus }, dead: boolean): WaitSignal | null { + if (dead || session.pid === null || session.pid === undefined) return 'exit'; + return signalForStatus(session.status); +} + +// ── Worker liveness for tmux-backed sessions ──────────────────────────────── +// +// `session.pid` is the LOCAL `tmux attach` client, not the worker. Codeman sets +// `remain-on-exit on` for every session it creates, so when the command inside the +// pane exits, tmux keeps the pane (`pane_dead=1`), the tmux session survives, the +// attach client keeps running and `pid` never goes null — no `exit` event is emitted +// and nothing in `Session` changes. Measured on a shell worker killed with `exit 42`: +// tmux reports `pane_dead=1 status=42` while Codeman reports `pid=309406 status=idle` +// and the DEFAULT wait answers `{signal:"idle", immediate:true}` in 0 ms for a corpse. +// So the liveness check has to ask the mux layer. `pid === null` still matters: it is +// the right (and only) answer for a direct-PTY session, which has no pane to ask about. +// +// Cost control, because `isPaneDead()` is a synchronous `execSync` and `/wait` is +// polled in a loop by design: +// 1. Only mux-backed sessions are probed at all. +// 2. Only requests that actually wait probe — a plain `POST .../input` (the browser's +// hot path, thousands per session) never touches tmux. +// 3. Results are cached per pane for PANE_DEATH_TTL_MS, so a poll loop cannot turn +// into one exec per request. +// 4. The while-blocked watcher is ONE timer per session no matter how many waiters +// are parked on it, and it exists only while at least one of them is. + +/** How long a pane-liveness probe is reused. Long enough to absorb a poll loop. */ +const PANE_DEATH_TTL_MS = 750; + +/** How often a session with a parked waiter is re-checked for a dead worker. */ +const PANE_DEATH_POLL_MS = 3_000; + +/** Bounded, because a 24h server churns through panes. */ +const paneDeathCache = new LRUMap({ maxSize: 256 }); + +/** One watcher per pane, refcounted by the waits currently parked on it. */ +const paneDeathWatchers = new Map(); + +type LivenessSession = { usesMux?: boolean; muxName?: string | null }; + +/** + * Whether the worker inside this session's tmux pane has exited. + * + * False for anything not tmux-backed (nothing to ask), and false when the probe is + * unavailable or throws — an unknown answer must never invent a death. + */ +function workerIsDead(mux: InfraPort['mux'], session: LivenessSession, now: number = Date.now()): boolean { + const muxName = session.usesMux === false ? null : session.muxName; + if (!muxName) return false; + // Defensive: `TerminalMultiplexer` declares it, but route-test doubles may not. + if (typeof mux?.isPaneDead !== 'function') return false; + + const cached = paneDeathCache.get(muxName); + if (cached && now - cached.at < PANE_DEATH_TTL_MS) return cached.dead; + + let dead = false; + try { + dead = mux.isPaneDead(muxName) === true; + } catch { + dead = false; + } + paneDeathCache.set(muxName, { dead, at: now }); + return dead; +} + +/** + * Release every waiter on a session whose worker has died, in the documented order. + * + * The same pair the PTY-exit listener and the delete path use, for the same reason: + * `until=exit` callers get their signal, everyone else gets `ended: true` instead of + * burning the rest of their timeout on feeds that will never produce anything. + */ +function releaseWaitersForDeadWorker(sessionId: string): void { + sessionWaits.notifySignal(sessionId, 'exit'); + sessionWaits.cancelAll(sessionId); +} + +/** + * While a wait is parked on a mux-backed session, poll for the worker dying. + * + * Without this, a worker that dies DURING a wait is invisible: no `exit` event fires + * (the attach client is still alive), no output arrives, and the caller blocks for its + * full timeout — the common orchestration case, "send a prompt and wait", where the + * worker crashes mid-turn. + * + * @returns a release function; call it in a `finally`, or the timer outlives the wait. + */ +function watchForDeadWorker(mux: InfraPort['mux'], session: LivenessSession, sessionId: string): () => void { + const muxName = session.usesMux === false ? null : session.muxName; + if (!muxName || typeof mux?.isPaneDead !== 'function') return () => {}; + + const existing = paneDeathWatchers.get(muxName); + if (existing) { + existing.refs++; + } else { + const timer = setInterval(() => { + if (!workerIsDead(mux, session)) return; + releaseWaitersForDeadWorker(sessionId); + }, PANE_DEATH_POLL_MS); + // Auxiliary to the waiter's own timer, which is deliberately NOT unref'd; this one + // must never be the reason the process stays up. + timer.unref(); + paneDeathWatchers.set(muxName, { timer, refs: 1 }); + } + + let released = false; + return () => { + if (released) return; + released = true; + const entry = paneDeathWatchers.get(muxName); + if (!entry) return; + entry.refs--; + if (entry.refs <= 0) { + clearInterval(entry.timer); + paneDeathWatchers.delete(muxName); + } + }; +} + +/** Test seam: pane-liveness state is module-level, so a suite must be able to reset it. */ +export function _resetPaneLivenessState(): void { + for (const entry of paneDeathWatchers.values()) clearInterval(entry.timer); + paneDeathWatchers.clear(); + paneDeathCache.clear(); +} + +/** Test seam: how many panes are currently being watched for a dead worker. */ +export function _paneDeathWatcherCount(): number { + return paneDeathWatchers.size; +} + +/** + * An `AbortController` that fires when the CLIENT goes away, and only then. + * + * Freeing an abandoned waiter matters because the documented pattern is a loop of + * short waits: `curl --max-time 30 ".../wait?timeout=600000"` abandons a live waiter + * every iteration until the cap is hit and an innocent session reports busy. Same for + * any proxy that cuts the connection. + * + * ⚠️ **It must listen on the RESPONSE, not the request.** `req.raw` emits `'close'` + * as soon as the request body has finished streaming, which on a POST is BEFORE the + * handler ever blocks — measured at +1ms with `aborted: false`, indistinguishable + * from a real hang-up at +0ms. Wiring the abort there cancels every send-and-wait + * instantly and silently kills the feature (it survives on GET only because a GET has + * no body to finish). `reply.raw` emits `'close'` both when the response completes + * and when the socket dies, and `writableFinished` is what tells those apart: true + * only if the response actually went out. The guard is load-bearing, not defensive. + * + * `app.inject()` never emits `'close'` at all, so this is only observable over real + * HTTP — which is why the regression test for it binds a port. + */ +function abortOnClientHangUp(reply: FastifyReply): AbortController { + const controller = new AbortController(); + reply.raw.on('close', () => { + if (!reply.raw.writableFinished) controller.abort(); + }); + return controller; +} + export function registerSessionRoutes( app: FastifyInstance, ctx: SessionPort & EventPort & ConfigPort & InfraPort & AuthPort @@ -863,9 +1104,9 @@ export function registerSessionRoutes( // ========== Send Input ========== - app.post('/api/sessions/:id/input', async (req) => { + app.post('/api/sessions/:id/input', async (req, reply) => { const { id } = req.params as { id: string }; - const { input, useMux, seq, clientId } = parseBody(SessionInputWithLimitSchema, req.body); + const { input, useMux, seq, clientId, wait, waitTimeout } = parseBody(SessionInputWithLimitSchema, req.body); const session = findSessionOrFail(ctx, id, req); const inputStr = String(input); @@ -876,27 +1117,102 @@ export function registerSessionRoutes( ); } + // Send-and-wait (agent orchestration). This has to be ONE endpoint rather than a + // POST followed by GET .../wait: between the write and the session flipping to + // `working` there is a window in which a separate wait sees the session still + // idle and returns instantly, reporting the PREVIOUS turn as this turn's answer. + // Registering the waiter before the write closes that window. + const wantsWait = + wait === true || (typeof wait === 'string' && wait.trim().length > 0) || (Array.isArray(wait) && wait.length > 0); + let until: readonly WaitSignal[] = []; + if (wantsWait) { + const resolved = resolveWaitSignals(wait === true ? undefined : wait, { mode: session.mode }); + if (resolved.error) return createErrorResponse(ApiErrorCode.INVALID_INPUT, resolved.error); + until = resolved.until; + } + // Reliable delivery (POST fallback when the WebSocket is down): a 2xx IS the // client's ACK, so a tagged duplicate redelivery must still return 200 but // skip the write. Untagged requests (curl/legacy) always apply. const tagged = typeof clientId === 'string' && typeof seq === 'number'; - if (tagged && !session.shouldApplyInput(clientId as string, seq as number)) { + const duplicate = tagged && !session.shouldApplyInput(clientId as string, seq as number); + if (duplicate && !wantsWait) { return {}; } + // Only a waiting request pays for the tmux probe: the browser's plain input path + // (thousands of calls per session) must stay exec-free. + const workerDead = wantsWait && workerIsDead(ctx.mux, session); + + const timeoutMs = clampWaitMs(waitTimeout ?? undefined); + // Same slot leak as the GET routes: a client that gives up mid-wait would + // otherwise hold a waiter for the full timeout. Response-side, always — see + // abortOnClientHangUp: on THIS route a request-side listener fires the moment the + // JSON body finishes streaming and aborts every send-and-wait before it starts. + const abort = abortOnClientHangUp(reply); + let waitPromise: Promise | null = null; + if (wantsWait) { + try { + waitPromise = sessionWaits.waitForSignal(id, { + until, + timeoutMs, + owner: ownerFor(req), + abortSignal: abort.signal, + // A FRESH delivery must not be satisfied by the state the session is already + // in: it is idle right now, which is precisely why we are typing at it. + // A DUPLICATE has no new turn coming, so it answers from the current state + // instead of blocking for a transition that already happened. + requireTransition: !duplicate, + currentSignal: duplicate ? currentSignalFor(session, workerDead) : undefined, + }); + } catch (err) { + if (err instanceof WaitCapacityError) { + // Nothing has been written yet, but `shouldApplyInput` already consumed the + // seq. Give it back or the caller's retry is rejected as a duplicate and the + // input is lost by the very mechanism meant to make delivery reliable. + if (tagged && !duplicate) session.forgetInputSeq(clientId as string, seq as number); + return waitCapacityResponse(err); + } + throw err; + } + } + const stopDeathWatch = wantsWait ? watchForDeadWorker(ctx.mux, session, id) : () => {}; + // Write input to PTY. Direct write is synchronous; writeViaMux // (tmux send-keys) is fire-and-forget to avoid blocking the HTTP response. - if (useMux) { + // + // Because the response has already been sent by then, a failure there is the + // one case the caller can never learn about — so the dedup bookkeeping is + // rolled back. Otherwise the seq stays recorded as applied and a retry, the + // very mechanism reliable delivery exists for, is rejected as a duplicate. + const undoOnFailure = () => { + if (tagged) session.forgetInputSeq(clientId as string, seq as number); + }; + + // Whether the bytes actually reached a write path. Only meaningful on the wait + // path (the fire-and-forget branches return before the response is built), and + // reported there instead of the old `!duplicate`: a PTY that has exited fails + // BOTH writes, and telling the caller "delivered, but it timed out" points it at + // the wrong recovery — wait longer, when the truth is "restart the worker". + let delivered = false; + + if (duplicate) { + // Redelivery of an already-applied input: skip the write, but still honor the + // wait, since the caller's question ("tell me when this settles") is unanswered. + } else if (useMux && waitPromise) { + // The response is already staying open for the wait, so the tmux write can be + // awaited here. This is the ONE path where a writeViaMux failure is observable. + const ok = await session.writeViaMux(inputStr).catch(() => false); + if (ok) { + delivered = true; + } else { + console.warn(`[Server] writeViaMux failed for session ${id}, falling back to direct write`); + delivered = session.write(inputStr); + if (!delivered) undoOnFailure(); + } + } else if (useMux) { // Fire-and-forget: don't block the HTTP response on a tmux child process. - // Fallback to a direct write on failure. - // - // Because the response has already been sent by then, a failure here is the - // one case the caller can never learn about — so the dedup bookkeeping is - // rolled back. Otherwise the seq stays recorded as applied and a retry, the - // very mechanism reliable delivery exists for, is rejected as a duplicate. - const undoOnFailure = () => { - if (tagged) session.forgetInputSeq(clientId as string, seq as number); - }; + // Fallback to a direct write on failure. Unchanged from before send-and-wait. session .writeViaMux(inputStr) .then((ok) => { @@ -911,11 +1227,209 @@ export function registerSessionRoutes( // Same rollback. NOT an error response, deliberately: a session can // legitimately have no PTY yet (created but not started), and callers have // always been able to write to one without a 4xx. - if (!session.write(inputStr) && tagged) { + delivered = session.write(inputStr); + if (!delivered && tagged) { session.forgetInputSeq(clientId as string, seq as number); } } - return {}; + + if (!waitPromise) return {}; + + try { + // `send-keys` SUCCEEDS against a dead pane — tmux is happy to write into a corpse + // — so a truthful `delivered` cannot come from the write's return value alone. + // This is the case the field exists for: "delivered, but it timed out" tells an + // agent to wait longer when the truth is "restart the worker". + if (delivered && workerDead) { + delivered = false; + // The bytes went nowhere, so the seq must not be recorded as applied or the + // caller's retry against a restarted worker is refused as a duplicate. + if (!duplicate) undoOnFailure(); + } + + // Nothing was written and nothing will be: no turn is coming, so blocking for the + // full timeout would only delay the caller's real recovery by up to ten minutes. + // Releasing the waiter also hands its slot back immediately. + const selfReleased = !delivered && !duplicate; + if (selfReleased) abort.abort(); + + const result = await waitPromise; + return { + success: true, + data: { + delivered, + duplicate, + status: session.status, + limitPaused: session.isLimitPaused, + // Identical `wait` object to the two GET endpoints, so one client helper + // reads all three, `timeoutMs` (post-clamp) included. + // + // `aborted` is the CLIENT-facing "you hung up, nobody is reading this", and + // by that definition it is unobservable — which is exactly what the API + // reference promises. The abort above is the server releasing its own waiter + // on a delivery that failed, and the client IS reading this response, so + // reporting `aborted: true` there would break that promise and hand an agent + // a second, contradictory reason for the same outcome. `delivered: false` + // already says what happened; `ended` says the wait was released early. + wait: { ...result, aborted: selfReleased ? false : result.aborted, until: [...until] }, + }, + }; + } finally { + stopDeathWatch(); + } + }); + + // ========== Wait For A Signal (agent orchestration) ========== + // + // A bounded long-poll: block until the session hits one of `until`, then answer. + // This exists because SSE is the only "tell me when" channel Codeman has, and an + // agent driving the API from a shell tool cannot hold a stream and parse events + // inline. See docs/agent-control-plan.md. + // + // A TIMEOUT IS A 200, not an error: callers are expected to loop over short waits + // (proxies such as `tailscale serve` cut idle connections), and turning every poll + // boundary into a 4xx would make that loop indistinguishable from a real failure. + + app.get('/api/sessions/:id/wait', async (req, reply) => { + const { id } = req.params as { id: string }; + const query = parseWaitQuery(SessionWaitQuerySchema, req.query, 'wait'); + const session = findSessionOrFail(ctx, id, req); + + // An agent polls this URL in a loop with identical parameters. Any intermediary + // applying heuristic freshness to the 200 would serve the stored `timedOut:true` + // body to the next iteration instantly, turning the loop into a busy spin that + // never observes the signal. + reply.header('Cache-Control', 'no-store'); + + // Shared with the `wait` field on POST .../input: unknown token is a 400, + // hook-only signals are rejected explicitly but dropped from the default. + const { until, error } = resolveWaitSignals(query.until, { mode: session.mode }); + if (error) return createErrorResponse(ApiErrorCode.INVALID_INPUT, error); + + // The value actually applied after clamping, echoed below: a caller that asked + // for 30 minutes and silently got 10 could not otherwise tell a poll boundary + // from a wedged worker, and would kill a session that was working fine. + const timeoutMs = clampWaitMs(query.timeout); + + // Free the waiter when the caller hangs up; the response can no longer be sent by + // then, so freeing the slot is the entire purpose. + const abort = abortOnClientHangUp(reply); + // A worker that dies while this request is parked emits nothing at all (the tmux + // attach client survives it), so a wait would otherwise run to its full timeout. + const stopDeathWatch = watchForDeadWorker(ctx.mux, session, id); + + try { + const result = await sessionWaits.waitForSignal(id, { + until, + timeoutMs, + owner: ownerFor(req), + abortSignal: abort.signal, + requireTransition: query.fresh === '1' || query.fresh === 'true', + // Read BEFORE awaiting: this is the state the caller is asking about. + currentSignal: currentSignalFor(session, workerIsDead(ctx.mux, session)), + }); + + return { + success: true, + data: { + sessionId: id, + // Post-wait status, so a caller that timed out still learns where things stand. + status: session.status, + // A session paused on a usage limit emits nothing until its reset, so a + // timeout here is expected rather than a stall worth retrying hard. + limitPaused: session.isLimitPaused, + // One shape across all three endpoints, so a single `is_done(resp)` helper + // works against any of them. `result.timeoutMs` is the value actually + // applied after clamping, which is what makes the clamp observable. + wait: { ...result, until: [...until] }, + }, + }; + } catch (err) { + if (err instanceof WaitCapacityError) return waitCapacityResponse(err); + throw err; + } finally { + stopDeathWatch(); + } + }); + + // ========== Wait For Output (agent orchestration) ========== + // + // The companion to /wait: block until a literal string appears in this session's + // output. Same 200-on-timeout contract. Fed by the `terminal` listener in + // session-listener-wiring.ts, so what this scans is byte-for-byte what the pane + // printed, ANSI stripped. + // + // ⚠️ A tmux repaint replays text already on screen, so `from=now` can match + // something printed before the request. Callers need a marker unique per call. + + app.get('/api/sessions/:id/wait-output', async (req, reply) => { + const { id } = req.params as { id: string }; + + // Reject `regex` loudly instead of ignoring it. Matching is deliberately literal + // (no ReDoS surface on a caller-supplied pattern over a live stream), and an agent + // that assumed otherwise would silently wait on the wrong thing. + if (req.query && typeof req.query === 'object' && 'regex' in req.query) { + return createErrorResponse( + ApiErrorCode.INVALID_INPUT, + 'regex is not supported; use match= (optionally with nocase=1)' + ); + } + + const query = parseWaitQuery(SessionWaitOutputQuerySchema, req.query, 'wait-output'); + const session = findSessionOrFail(ctx, id, req); + + // Same reason as /wait: this URL is polled in a loop with identical parameters. + reply.header('Cache-Control', 'no-store'); + + const timeoutMs = clampWaitMs(query.timeout); + const abort = abortOnClientHangUp(reply); + const owner = ownerFor(req); + // Output waiters are the ones a dead worker strands hardest: the feed simply stops. + const stopDeathWatch = watchForDeadWorker(ctx.mux, session, id); + + try { + // Check the cap BEFORE touching the buffer. `session.terminalBuffer` is + // `BufferAccumulator.value`, which joins the WHOLE accumulator (up to 32MB) + // before the slice below takes its tail — so a request that is going to be + // rejected anyway must not pay for a full materialization first, or the cap + // provides no backpressure at all against a `from=buffer` loop. + sessionWaits.assertCapacity(id, owner); + + // `from=buffer` scans what already scrolled past before blocking. Bounded to a + // tail: the buffer runs to 32MB and this is a per-request ANSI strip. + let initialText: string | undefined; + if (query.from === 'buffer') { + const buffer = session.terminalBuffer; + initialText = + buffer.length > MAX_BUFFER_SCAN_BYTES ? buffer.slice(buffer.length - MAX_BUFFER_SCAN_BYTES) : buffer; + } + + const result = await sessionWaits.waitForOutput(id, { + match: query.match, + nocase: query.nocase === '1' || query.nocase === 'true', + timeoutMs, + owner, + abortSignal: abort.signal, + initialText, + }); + + return { + success: true, + data: { + sessionId: id, + status: session.status, + limitPaused: session.isLimitPaused, + // Same envelope as /wait; this one carries `matched`/`snippet`/`match` + // where the signal wait carries `signal`/`until`. + wait: { ...result, match: query.match }, + }, + }; + } catch (err) { + if (err instanceof WaitCapacityError) return waitCapacityResponse(err); + throw err; + } finally { + stopDeathWatch(); + } }); // ========== Send Named Key (tmux send-keys -H) ========== diff --git a/src/web/schemas.ts b/src/web/schemas.ts index 19879feb..8282157f 100644 --- a/src/web/schemas.ts +++ b/src/web/schemas.ts @@ -17,6 +17,7 @@ import { MIN_TERMINAL_SCROLLBACK_LINES, } from '../config/terminal-history.js'; import { MAX_EDITABLE_BYTES } from '../config/file-editing.js'; +import { MIN_MATCH_LENGTH, MAX_MATCH_LENGTH } from '../config/agent-wait.js'; // ========== Path Validation ========== @@ -911,6 +912,66 @@ export const SessionInputWithLimitSchema = z.object({ // unset rather than sending null. See docs/reliable-input-delivery.md. seq: z.number().int().nonnegative().optional(), clientId: z.string().max(128).optional(), + // Send-and-wait (agent orchestration): `true` for the default signal set, or the + // same grammar as `GET .../wait` — a comma string or an array of signals. Absent + // means the historical fire-and-forget behavior, byte for byte. + // + // `.nullish()`, not `.optional()`: a third-party caller building the body with + // JSON.stringify keeps an explicit null on the wire, and `.optional()` rejects it + // with INVALID_INPUT. That gotcha has shipped as a real bug twice. + wait: z.union([z.boolean(), z.string().max(120), z.array(z.string().max(120)).max(8)]).nullish(), + // Unbounded above: the effective value is clamped to MAX_WAIT_MS server-side and + // returned as `data.wait.timeoutMs`, so a caller that asks for 24h sees what it + // actually got. A `.max()` here would turn the same documented clamp into a 400 for + // large-enough guesses, which is the one behaviour an agent cannot predict. + waitTimeout: z.number().int().positive().nullish(), +}); + +/** + * Query validation for `GET /api/sessions/:id/wait` (agent wait primitives). + * + * Everything arrives as a string. `timeout` is coerced and bounded here, then + * clamped again to the operator's ceiling by `clampWaitMs()` — the schema bound + * only keeps an absurd number out of the arithmetic. A non-numeric `timeout` is a + * 400 rather than a silent fallback, so an agent never believes it asked for a + * longer wait than it got; the value actually applied comes back as + * `data.wait.timeoutMs`, which is what makes the clamp observable. `until` is + * parsed by `parseWaitSignals()`, which reports unknown tokens instead of + * dropping them. + * + * `until` accepts an ARRAY as well as the comma string: `?until=stop&until=exit` + * is how most HTTP clients express a list, Fastify's query parser delivers a + * repeated parameter as an array, and `parseWaitSignals()` has always handled + * both. Rejecting the repeated form left that branch unreachable and 400'd the + * more natural spelling. + */ +export const SessionWaitQuerySchema = z.object({ + until: z.union([z.string().max(120), z.array(z.string().max(120)).max(8)]).optional(), + // No upper bound on purpose. The contract is "clamped to [MIN_WAIT_MS, MAX_WAIT_MS]", + // and a `.max()` here contradicted it: `timeout=99999999` was a 400 mid-fan-out while + // `timeout=600001` was silently clamped, so the same documented rule produced two + // different outcomes depending on how big the caller's guess was. `clampWaitMs()` + // bounds every finite value, and `.int()` still rejects `Infinity`/`1e999` and junk. + timeout: z.coerce.number().int().positive().optional(), + fresh: z.enum(['0', '1', 'true', 'false']).optional(), +}); + +/** + * Query validation for `GET /api/sessions/:id/wait-output`. + * + * `match` is a LITERAL substring, never a pattern: `search-service.ts` avoids regex + * so there is no ReDoS surface, and this endpoint is more exposed still (the pattern + * would be caller-supplied and the input is a live stream). The length bound is a + * second reason the carry buffer stays small. The route separately rejects a `regex` + * parameter outright rather than ignoring it. + */ +export const SessionWaitOutputQuerySchema = z.object({ + match: z.string().min(MIN_MATCH_LENGTH).max(MAX_MATCH_LENGTH), + nocase: z.enum(['0', '1', 'true', 'false']).optional(), + from: z.enum(['now', 'buffer']).optional(), + // Unbounded above for the same reason as SessionWaitQuerySchema.timeout: clamping is + // the documented contract, so a large value must clamp rather than 400. + timeout: z.coerce.number().int().positive().optional(), }); // ========== Session Mutation Routes ========== diff --git a/src/web/server.ts b/src/web/server.ts index 74b9c71d..3d7f1921 100644 --- a/src/web/server.ts +++ b/src/web/server.ts @@ -85,6 +85,7 @@ import { attachSessionListeners, detachSessionListeners, } from './session-listener-wiring.js'; +import { sessionWaits } from './session-wait-registry.js'; import { wireRespawnListeners, setupTimedRespawn, @@ -1247,6 +1248,16 @@ export class WebServer extends EventEmitter { } } + // Release anything blocked on this session, in the documented order: 'exit' + // first so an until=exit caller gets its signal, then cancelAll so everyone + // else resolves with ended:true instead of timing out. + // + // The 'exit' here is NOT redundant with the PTY-exit listener: listeners are + // detached a few lines above, before `session.stop()`, so on a delete the + // session's own exit event never reaches the registry. + sessionWaits.notifySignal(sessionId, 'exit'); + sessionWaits.cancelAll(sessionId); + this.broadcast(SseEvent.SessionDeleted, { id: sessionId }); } @@ -2845,6 +2856,11 @@ export class WebServer extends EventEmitter { // Gracefully close all SSE connections and clear batching state this.sse.stop(); + // Release every pending long-poll waiter. Their timers are deliberately not + // unref'd (an unref'd timer can let the process exit mid-wait and strand the + // response), so without this a 10-minute wait holds shutdown open. + sessionWaits.cancelEverything(); + this.lastRecordedTokens.clear(); // Stop multiplexer and flush pending saves diff --git a/src/web/session-listener-wiring.ts b/src/web/session-listener-wiring.ts index 3f040e7d..af95e853 100644 --- a/src/web/session-listener-wiring.ts +++ b/src/web/session-listener-wiring.ts @@ -27,6 +27,7 @@ import type { RalphStatusBlock, CircuitBreakerStatus } from '../types.js'; import { SseEvent } from './sse-events.js'; import { getLifecycleLog } from '../session-lifecycle-log.js'; import { fileStreamManager } from '../file-stream-manager.js'; +import { sessionWaits } from './session-wait-registry.js'; /** Stored listener references for session cleanup (prevents memory leaks) */ export interface SessionListenerRefs { @@ -92,6 +93,9 @@ export function createSessionListeners(session: Session, deps: SessionListenerDe /** Batches PTY output → broadcasts `session:terminal` at 16-50ms intervals */ terminal: (data) => { + // Feeds `GET /api/sessions/:id/wait-output`. No-ops with a single Map lookup + // when nothing is waiting, which is the case on virtually every chunk. + sessionWaits.notifyOutput(session.id, data); deps.batchTerminalData(session.id, data); }, @@ -137,6 +141,28 @@ export function createSessionListeners(session: Session, deps: SessionListenerDe /** Broadcasts `session:exit` + `session:updated` — PTY process exited; cleans up respawn, timers, listeners */ exit: (code) => { + // Before anything that can throw: a caller blocked on this session must learn + // the process died rather than sit until its timeout. + // + // Both halves are required, in this order — the same pair `_doCleanupSession` + // uses on the delete path, for the same reason. `notifySignal` resolves ONLY + // waiters that asked for `exit`; everyone else (`until=working`, `until=stop`, + // every wait-output) would keep a slot in the process-wide pool until their + // timeout, on a session whose feeds this very handler is about to tear down: + // `removeSessionListenerRefs` below detaches the `terminal` listener that is + // the only input to `notifyOutput`, and the `idle`/`working` listeners with it. + // Nothing can reach those waiters afterwards, so holding them is a guaranteed + // ten-minute lie. `cancelAll` answers them `ended: true`, which the plan's §3.6 + // specifies for exactly this case ("Never hang"). + // + // Safe against the respawn cycle: a respawn writes `/clear` + a kickstart + // prompt through the mux and never restarts the PTY, so it emits no `exit` and + // cannot cancel an orchestrating agent's wait. And for an agent driving a + // worker this is the right trade even when the PTY exit was only a tmux + // DETACH: `ended` means "re-check and re-issue", one extra round trip, versus + // burning the caller's entire timeout learning nothing. + sessionWaits.notifySignal(session.id, 'exit'); + sessionWaits.cancelAll(session.id); getLifecycleLog().log({ event: 'exit', sessionId: session.id, @@ -187,6 +213,7 @@ export function createSessionListeners(session: Session, deps: SessionListenerDe /** Broadcasts `session:working` — Claude started processing */ working: () => { + sessionWaits.notifySignal(session.id, 'working'); deps.broadcast(SseEvent.SessionWorking, { id: session.id }); const tracker = deps.getRunSummaryTracker(session.id); if (tracker) { @@ -197,6 +224,7 @@ export function createSessionListeners(session: Session, deps: SessionListenerDe /** Broadcasts `session:idle` — Claude finished processing, waiting for input */ idle: () => { + sessionWaits.notifySignal(session.id, 'idle'); deps.broadcast(SseEvent.SessionIdle, { id: session.id }); deps.broadcastSessionStateDebounced(session.id); const tracker = deps.getRunSummaryTracker(session.id); diff --git a/src/web/session-wait-registry.ts b/src/web/session-wait-registry.ts new file mode 100644 index 00000000..6675f2e8 --- /dev/null +++ b/src/web/session-wait-registry.ts @@ -0,0 +1,1089 @@ +/** + * @fileoverview Blocking-wait registry backing the agent wait primitives. + * + * Codeman's only "tell me when" channel today is SSE, which an agent driving the + * API from a shell tool cannot practically consume (it would have to hold a + * streaming connection and parse events inline). This registry is the piece that + * lets a request *block* until something happens instead, so an orchestrating + * agent can `curl` and wait. Plan: `docs/agent-control-plan.md`. + * + * Two waiter kinds: + * - **signal waiters** resolve on the first of a requested set of lifecycle + * signals (`idle`, `working`, `stop`, `blocked`, `exit`). + * - **output waiters** resolve when a literal string appears in a session's + * ANSI-stripped output. + * + * ## What "ANSI-stripped output" means here + * + * `normalizeForMatch()` defines it, and it is a superset of `stripAnsi()`: that helper + * knows three escape families and leaves the rest, which is enough for `ESC ( B` from a + * stock bash prompt to break `match=tnode:` on a prompt that reads `tnode:`. Everything + * downstream derives from one call to it, so the carry, the haystack and the snippet + * window are all the same text. The returned `snippet` is then RENDERED from that window + * (control bytes removed, blank runs collapsed) and so is not a byte-for-byte quotation + * of what was matched — see `findMatch`. + * + * ## Signal provenance (why `stop` is the good one) + * + * `idle` / `working` / `exit` come from `Session`'s own events (heuristic: output + * stabilization plus prompt detection), so `idle` can flap mid-turn when a spinner + * pauses. `stop` and `blocked` come from Claude Code hooks (`POST /api/hook-event`) + * and are definitive. Only `claude` mode installs those hooks, so `stop` and + * `blocked` never fire for anything else: callers must be rejected up front rather + * than left to hit the timeout. `hooksAvailableForMode()` keeps that rule in one + * place, and it is keyed on the MODE rather than on `isExternalCliMode()` because + * `shell` is not an external CLI yet installs no hooks either. + * + * ## Ordering contract for the wiring + * + * A session that exits must `notifySignal(id, 'exit')` BEFORE `cancelAll(id)`, so + * a caller waiting on `exit` gets `signal: 'exit'` rather than `ended: true`. + * + * ## Lifetime discipline (24h sessions) + * + * Every waiter owns exactly one timer, cleared on resolve, and waiter sets are + * deleted when they empty. Session teardown must call `cancelAll()` and shutdown + * `stop()`: timers are deliberately NOT unref'd, because an unref'd timer can let + * the process exit mid-wait and strand the HTTP response. + * + * Two things keep a slot from outliving its caller: + * - `abortSignal`, so a client that hangs up frees its waiter immediately instead of + * holding one for the rest of its timeout. Routes wire it to `req.raw.on('close')`. + * - the latched stopped flag, so a request that lands in the window between shutdown + * starting and the HTTP server actually closing cannot register a fresh waiter that + * nothing will ever cancel. + * + * This module holds no IO and no `Session` reference, which is what keeps it + * unit-testable in isolation. + */ + +import { stripAnsi } from '../utils/index.js'; +import { + MAX_WAITERS_PER_SESSION, + MAX_WAITERS_PER_OWNER, + MAX_WAITERS_TOTAL, + MAX_SNIPPET_CONTEXT, +} from '../config/agent-wait.js'; +import type { SessionMode, SessionStatus } from '../types.js'; + +// ─── Signals ───────────────────────────────────────────────────────────────── + +/** A lifecycle signal a caller can block on. */ +export type WaitSignal = 'idle' | 'working' | 'stop' | 'blocked' | 'exit'; + +/** Every valid signal, in documentation order. */ +export const WAIT_SIGNALS: readonly WaitSignal[] = ['idle', 'working', 'stop', 'blocked', 'exit']; + +/** + * Applied when a caller omits `until`. `stop` first because it is the definitive + * end-of-turn signal, `idle` as the fallback for sessions that emit no hooks, and + * `exit` so a worker that CRASHES resolves the wait promptly instead of burning + * the caller's whole timeout on something that can no longer happen. + */ +export const DEFAULT_WAIT_SIGNALS: readonly WaitSignal[] = ['stop', 'idle', 'exit']; + +const SIGNAL_SET = new Set(WAIT_SIGNALS); + +/** Result of parsing a caller-supplied `until` value. */ +export interface ParsedWaitSignals { + /** Valid signals, deduped, in the order given. */ + signals: WaitSignal[]; + /** Tokens that are not signals. Non-empty means the caller should get a 400. */ + invalid: string[]; +} + +/** + * Parse an `until` query value: a comma-separated string, an array of strings, or + * absent. Invalid tokens are REPORTED rather than dropped: silently falling back + * to the default would leave an agent believing it is waiting for `stop` when a + * typo means it is waiting for something else entirely. + * + * @param raw - `"stop,idle"`, `["stop","idle"]`, or undefined + * @returns valid signals plus any unrecognized tokens + */ +export function parseWaitSignals(raw: unknown): ParsedWaitSignals { + const tokens: string[] = []; + if (typeof raw === 'string') { + tokens.push(...raw.split(',')); + } else if (Array.isArray(raw)) { + for (const entry of raw) { + if (typeof entry === 'string') tokens.push(...entry.split(',')); + } + } + + const signals: WaitSignal[] = []; + const invalid: string[] = []; + const seen = new Set(); + for (const token of tokens) { + const value = token.trim().toLowerCase(); + if (!value) continue; + if (!SIGNAL_SET.has(value)) { + if (!invalid.includes(value)) invalid.push(value); + continue; + } + if (seen.has(value)) continue; + seen.add(value); + signals.push(value as WaitSignal); + } + return { signals, invalid }; +} + +/** + * The signal a session is *already* emitting, given its status, so a wait can + * resolve immediately instead of hanging until something changes. + * + * ⚠️ **What this CANNOT tell you: whether the session is alive.** It sees only the + * four-value `SessionStatus` enum, and that enum does not distinguish a finished + * worker from a dead one. Both PTY `onExit` handlers in `session.ts` (Claude + * interactive and shell) park the session at `'idle'`, and a PTY exit does not remove + * the session from the map, so a crashed worker reports `'idle'` indefinitely and this + * function will answer `'idle'` for it forever. `'stopped'` is set only on a spawn + * failure and by `stop(killMux:true)` (which deletes the session moments later), so in + * practice the `'stopped' → 'exit'` arm almost never fires for a session that merely + * died. + * + * Read the return value as "what state is it in", never as "is it still running". A + * caller that must tell a completed turn from a corpse has to consult liveness itself + * (`session.pid`, which is null after a PTY exit) and cannot get it from here. The + * live `exit` signal is still delivered by `notifySignal(id, 'exit')` at the moment + * the PTY dies; this function is only about the state a LATER caller finds. + * + * `stopped` and `error` both map to `exit` because both mean "not running", so a + * caller that does catch one of them does not block. `blocked` is deliberately + * underivable here, it exists only as a hook event until it becomes a real + * `SessionStatus` (see the deferred Part 3 in the plan). + */ +export function signalForStatus(status: SessionStatus): WaitSignal | null { + switch (status) { + case 'idle': + return 'idle'; + case 'busy': + return 'working'; + case 'stopped': + case 'error': + return 'exit'; + default: + return null; + } +} + +/** Signals that arrive only via Claude Code hooks, so only `claude` mode can emit them. */ +const HOOK_ONLY_SIGNALS: readonly WaitSignal[] = ['stop', 'blocked']; + +/** + * Whether a session in this mode ever POSTs Codeman hook events, and therefore + * whether `stop` / `blocked` can ever fire for it. + * + * True for `claude` and nothing else. The tempting predicate is + * `!isExternalCliMode(mode)`, and it is WRONG: that helper covers only + * opencode/codex/gemini/antigravity, so `shell` falls through it — and a shell session + * is a plain bash PTY with no Claude Code and no hooks installed. `until=stop` on one + * was accepted and then blocked for the caller's whole timeout, which is precisely the + * infinite-wait-dressed-as-a-timeout this guard exists to prevent. + */ +export function hooksAvailableForMode(mode: SessionMode): boolean { + return mode === 'claude'; +} + +/** Outcome of resolving a caller-supplied wait target against a session's mode. */ +export interface ResolvedWaitSignals { + /** Signals to actually wait on. Empty when `error` is set. */ + until: WaitSignal[]; + /** Caller-facing 400 message, or null when the request is usable. */ + error: string | null; +} + +/** + * Turn a raw `until` / `wait` value into the set to wait on, applying both rules + * every wait endpoint shares: + * + * 1. An unknown token is an ERROR, never a silent fallback to the default. + * 2. Hook-only signals are rejected when asked for EXPLICITLY on a mode that emits + * no hooks, but merely dropped from the DEFAULT set: omitting the parameter must + * never 400. + * + * Shared by `GET .../wait` and the `wait` field on `POST .../input` so the two can + * not drift; the second-guessing that produces is worse than the duplication. + * + * @param raw - the caller's value (comma string, array, `true` for "the default") + * @param options - `mode` decides whether the hook-only signals are available, and + * names the mode in the error message so the caller can see why + */ +export function resolveWaitSignals(raw: unknown, options: { mode: SessionMode }): ResolvedWaitSignals { + const parsed = parseWaitSignals(raw); + if (parsed.invalid.length > 0) { + return { + until: [], + error: `Unknown wait signal(s): ${parsed.invalid.join(', ')}. Valid: ${WAIT_SIGNALS.join(', ')}`, + }; + } + + const unsupported = new Set(hooksAvailableForMode(options.mode) ? [] : HOOK_ONLY_SIGNALS); + + if (parsed.signals.length === 0) { + return { until: DEFAULT_WAIT_SIGNALS.filter((signal) => !unsupported.has(signal)), error: null }; + } + + const rejected = parsed.signals.filter((signal) => unsupported.has(signal)); + if (rejected.length > 0) { + return { + until: [], + error: `Signal(s) ${rejected.join(', ')} never fire for ${options.mode} sessions (no Claude Code hooks). Use idle or exit.`, + }; + } + return { until: parsed.signals, error: null }; +} + +// ─── Results ───────────────────────────────────────────────────────────────── + +/** Fields every wait result carries, whichever kind it is. */ +interface WaitResultBase { + /** True when the wait hit its timeout. */ + timedOut: boolean; + /** True when the answer came from state already present at call time. */ + immediate: boolean; + /** True when the session went away (deleted / torn down) before the wait resolved. */ + ended: boolean; + /** + * True when the caller's `abortSignal` fired: the client hung up, so nobody is + * reading this result. `ended` is set alongside it, since the wait did end without + * an answer. False on every other path. + */ + aborted: boolean; + /** Wall-clock ms spent waiting (0 for an immediate resolve). */ + waitedMs: number; + /** + * The timeout actually applied, after clamping. Echoed because a caller that asked + * for 30 minutes and silently got 600s otherwise reads the timeout as a stalled + * worker and kills a session that was working fine. + */ + timeoutMs: number; +} + +/** Outcome of a signal wait. A timeout is a normal outcome, never an error. */ +export interface SignalWaitResult extends WaitResultBase { + /** The signal that fired, or null if the wait ended without one. */ + signal: WaitSignal | null; +} + +/** Outcome of an output wait. */ +export interface OutputWaitResult extends WaitResultBase { + matched: boolean; + /** Bounded window of output around the match, or null if nothing matched. */ + snippet: string | null; +} + +/** + * Thrown when a waiter would exceed a configured cap. + * + * `scope` is what the route needs to answer honestly: `'session'` is the caller's own + * session being oversubscribed (409 SESSION_BUSY), while `'owner'` and `'total'` are + * caps the caller may have no part in and cannot fix by switching sessions, so those + * map to 429 RATE_LIMITED. Reporting a process-wide cap as SESSION_BUSY tells the + * caller the wrong session is at fault. + */ +export class WaitCapacityError extends Error { + override readonly name = 'WaitCapacityError'; + constructor( + message: string, + /** Which cap was hit, so the route can say something useful. */ + readonly scope: 'session' | 'owner' | 'total' + ) { + super(message); + } +} + +// ─── Options ───────────────────────────────────────────────────────────────── + +/** Options shared by both wait kinds. */ +interface WaitOptionsBase { + /** Timeout in ms. Callers pass the CLAMPED value; it is echoed back in the result. */ + timeoutMs: number; + /** + * Owner to charge this waiter to (multi-user mode). Omit in single-user mode: the + * per-owner cap applies only when this is set, so leaving it undefined behaves + * exactly as if the cap did not exist. + */ + owner?: string; + /** + * Fires when the caller goes away (routes wire this to `req.raw.on('close')`). + * The waiter is removed and its timer cleared at once, resolving `ended: true, + * aborted: true`. Freeing the slot is the entire point: the response can no longer + * be sent, so holding one for the rest of the timeout only denies the pool to + * someone else. An already-aborted signal registers no waiter at all. + */ + abortSignal?: AbortSignal; +} + +export interface SignalWaitOptions extends WaitOptionsBase { + /** Signals to wait for; the first to fire wins. An empty set can only time out. */ + until: readonly WaitSignal[]; + /** + * When true, ignore `currentSignal` and require an actual transition. This is + * the `fresh=1` behavior: "tell me about the NEXT one", not "is it already so". + */ + requireTransition?: boolean; + /** The session's current signal (from `signalForStatus`), if known. */ + currentSignal?: WaitSignal | null; +} + +export interface OutputWaitOptions extends WaitOptionsBase { + /** Literal substring to look for. No regex: see the note on `waitForOutput`. */ + match: string; + /** Case-insensitive compare. */ + nocase?: boolean; + /** + * Existing buffered output to scan before waiting (the `from=buffer` mode). + * Callers should pass a BOUNDED slice: a session's text buffer can be tens of + * megabytes and this is scanned synchronously. `assertCapacity()` runs BEFORE this + * is touched, so a request that cannot get a slot never pays for the scan. + */ + initialText?: string; +} + +export interface SessionWaitRegistryOptions { + maxWaitersPerSession?: number; + maxWaitersPerOwner?: number; + maxWaitersTotal?: number; +} + +// ─── Internal waiter records ───────────────────────────────────────────────── + +/** Bookkeeping every waiter carries, whichever kind it is. */ +interface WaiterBase { + startedAt: number; + timer: NodeJS.Timeout; + /** Effective timeout, echoed back in the result. */ + timeoutMs: number; + /** Owner this waiter is charged to, if any. */ + owner?: string; + /** Detaches the abort listener; called on every removal path. */ + detachAbort?: () => void; +} + +interface SignalWaiter extends WaiterBase { + until: Set; + settle: (result: SignalWaitResult) => void; +} + +interface OutputWaiter extends WaiterBase { + /** Original-case needle, kept for reporting. */ + needle: string; + /** Comparison form of the needle (lowercased when nocase). */ + needleCmp: string; + nocase: boolean; + /** Tail of previously scanned text, so a match can straddle two chunks. */ + carry: string; + settle: (result: OutputWaitResult) => void; +} + +// ─── Registry ──────────────────────────────────────────────────────────────── + +/** + * Holds pending waiters keyed by session id. + * + * One instance is shared process-wide (`sessionWaits` below); tests construct + * their own so no state leaks between cases. + */ +export class SessionWaitRegistry { + private readonly signalWaiters = new Map>(); + private readonly outputWaiters = new Map>(); + /** + * Live waiter count per owner. A derived count would mean walking every waiter in + * the process on each registration, so it is maintained incrementally instead: + * incremented once at registration, decremented once in the remove* helpers, which + * are the only paths that take a waiter out of a set. + */ + private readonly ownerCounts = new Map(); + /** + * Trailing bytes of a PTY chunk that look like the START of an ANSI sequence whose + * terminator has not arrived yet, held back until the next chunk completes it. Keyed + * by session because every output waiter on a session sees the same stream, which + * keeps `notifyOutput` at ONE `stripAnsi` per chunk rather than one per waiter. + * Deleted with the session's waiter set. + */ + private readonly pendingAnsi = new Map(); + private readonly maxPerSession: number; + private readonly maxPerOwner: number; + private readonly maxTotal: number; + /** Latched by `stop()` / `cancelEverything()`. Never cleared: shutdown is one-way. */ + private stopped = false; + + constructor(options: SessionWaitRegistryOptions = {}) { + this.maxPerSession = options.maxWaitersPerSession ?? MAX_WAITERS_PER_SESSION; + this.maxPerOwner = options.maxWaitersPerOwner ?? MAX_WAITERS_PER_OWNER; + this.maxTotal = options.maxWaitersTotal ?? MAX_WAITERS_TOTAL; + } + + // ── Counts (also the cap accounting) ── + + /** Pending signal waiters for a session. */ + signalWaiterCount(sessionId: string): number { + return this.signalWaiters.get(sessionId)?.size ?? 0; + } + + /** Pending output waiters for a session. */ + outputWaiterCount(sessionId: string): number { + return this.outputWaiters.get(sessionId)?.size ?? 0; + } + + /** Pending waiters of both kinds for a session. */ + waiterCount(sessionId: string): number { + return this.signalWaiterCount(sessionId) + this.outputWaiterCount(sessionId); + } + + /** Pending waiters of both kinds across every session. */ + totalWaiterCount(): number { + let total = 0; + for (const set of this.signalWaiters.values()) total += set.size; + for (const set of this.outputWaiters.values()) total += set.size; + return total; + } + + /** Pending waiters of both kinds charged to one owner. */ + ownerWaiterCount(owner: string): number { + return this.ownerCounts.get(owner) ?? 0; + } + + /** True once `stop()` / `cancelEverything()` has run: no new waiter can register. */ + get isStopped(): boolean { + return this.stopped; + } + + /** + * Throw if registering one more waiter would exceed a cap. + * + * PUBLIC so a route can check BEFORE doing expensive setup work. `/wait-output`'s + * `from=buffer` mode is the case that matters: reading `session.terminalBuffer` + * joins the whole 32MB accumulator, and paying that for a request that is about to + * be refused turns the cap into an amplifier instead of a protection. Both + * `waitForSignal` and `waitForOutput` call it themselves too, so a caller that skips + * the pre-check is still bounded. + * + * @throws {WaitCapacityError} naming the cap that was hit in `scope`. + */ + assertCapacity(sessionId: string, owner?: string): void { + if (this.totalWaiterCount() >= this.maxTotal) { + throw new WaitCapacityError(`Too many concurrent waits (scope: total, max ${this.maxTotal})`, 'total'); + } + if (owner !== undefined && this.ownerWaiterCount(owner) >= this.maxPerOwner) { + throw new WaitCapacityError( + `Too many concurrent waits for this user (scope: owner, max ${this.maxPerOwner})`, + 'owner' + ); + } + if (this.waiterCount(sessionId) >= this.maxPerSession) { + throw new WaitCapacityError( + `Too many concurrent waits on this session (scope: session, max ${this.maxPerSession})`, + 'session' + ); + } + } + + private chargeOwner(owner: string | undefined): void { + if (owner === undefined) return; + this.ownerCounts.set(owner, (this.ownerCounts.get(owner) ?? 0) + 1); + } + + private releaseOwner(owner: string | undefined): void { + if (owner === undefined) return; + const next = (this.ownerCounts.get(owner) ?? 0) - 1; + if (next > 0) this.ownerCounts.set(owner, next); + else this.ownerCounts.delete(owner); + } + + /** + * Wire a caller's abort signal to `onAbort`, returning the detach function. + * Detaching matters even though request signals are short-lived: without it a + * caller reusing one controller across several waits accumulates listeners on it. + */ + private attachAbort(signal: AbortSignal | undefined, onAbort: () => void): (() => void) | undefined { + if (!signal) return undefined; + signal.addEventListener('abort', onAbort, { once: true }); + return () => signal.removeEventListener('abort', onAbort); + } + + // ── Signal waits ── + + /** + * Block until one of `until` fires, the timeout elapses, or the session ends. + * + * Resolves immediately (`immediate: true`) when the session's `currentSignal` is + * already in `until` and `requireTransition` is not set. + * + * @throws {WaitCapacityError} when a waiter cap is already reached. + */ + waitForSignal(sessionId: string, options: SignalWaitOptions): Promise { + const until = new Set(options.until); + const base = { signal: null, timedOut: false, immediate: false, waitedMs: 0, timeoutMs: options.timeoutMs }; + + // Nobody is listening / nothing can ever resolve: answer without taking a slot, + // and without a capacity throw the caller could not act on either way. + if (options.abortSignal?.aborted) return Promise.resolve({ ...base, ended: true, aborted: true }); + if (this.stopped) return Promise.resolve({ ...base, ended: true, aborted: false }); + + if (!options.requireTransition && options.currentSignal && until.has(options.currentSignal)) { + return Promise.resolve({ + signal: options.currentSignal, + timedOut: false, + immediate: true, + ended: false, + aborted: false, + waitedMs: 0, + timeoutMs: options.timeoutMs, + }); + } + + this.assertCapacity(sessionId, options.owner); + + return new Promise((resolve) => { + let set = this.signalWaiters.get(sessionId); + if (!set) { + set = new Set(); + this.signalWaiters.set(sessionId, set); + } + + const waiter: SignalWaiter = { + until, + startedAt: Date.now(), + timeoutMs: options.timeoutMs, + owner: options.owner, + timer: setTimeout(() => { + this.removeSignalWaiter(sessionId, waiter); + resolve({ + signal: null, + timedOut: true, + immediate: false, + ended: false, + aborted: false, + waitedMs: Date.now() - waiter.startedAt, + timeoutMs: waiter.timeoutMs, + }); + }, options.timeoutMs), + settle: resolve, + }; + set.add(waiter); + this.chargeOwner(options.owner); + waiter.detachAbort = this.attachAbort(options.abortSignal, () => { + this.removeSignalWaiter(sessionId, waiter); + resolve({ + signal: null, + timedOut: false, + immediate: false, + ended: true, + aborted: true, + waitedMs: Date.now() - waiter.startedAt, + timeoutMs: waiter.timeoutMs, + }); + }); + }); + } + + /** + * Deliver a lifecycle signal. Every waiter for this session that asked for it + * resolves; the rest keep waiting. + * + * @returns how many waiters were resolved (0 is the overwhelmingly common case, + * which is why this is safe to call from the session event hot path). + */ + notifySignal(sessionId: string, signal: WaitSignal): number { + const set = this.signalWaiters.get(sessionId); + if (!set || set.size === 0) return 0; + + // Snapshot: settling mutates the set. + let resolved = 0; + for (const waiter of [...set]) { + if (!waiter.until.has(signal)) continue; + this.removeSignalWaiter(sessionId, waiter); + waiter.settle({ + signal, + timedOut: false, + immediate: false, + ended: false, + aborted: false, + waitedMs: Date.now() - waiter.startedAt, + timeoutMs: waiter.timeoutMs, + }); + resolved++; + } + return resolved; + } + + private removeSignalWaiter(sessionId: string, waiter: SignalWaiter): void { + clearTimeout(waiter.timer); + waiter.detachAbort?.(); + const set = this.signalWaiters.get(sessionId); + if (!set) return; + // Guard on the delete: a second removal (abort racing a resolve) must not + // decrement the owner count twice and hand out a slot that is still in use. + if (!set.delete(waiter)) return; + this.releaseOwner(waiter.owner); + if (set.size === 0) this.signalWaiters.delete(sessionId); + } + + // ── Output waits ── + + /** + * Block until `match` appears in the session's output. + * + * **Literal matching only, deliberately.** `search-service.ts` avoids regex so + * there is no ReDoS surface, and this endpoint is more exposed still: the + * pattern is caller-supplied and the input is a live stream. herdr can offer + * `--regex` because Rust's regex crate is linear-time with no backtracking; + * JavaScript's `RegExp` backtracks. + * + * ⚠️ "New output" is not the same as "output produced after you asked". tmux + * REPAINTS the visible screen (on attach, resize, or a TUI redraw), and a repaint + * arrives as ordinary `terminal` data, so text that was already on screen can + * match a `from=now` wait. Observed live: a marker echoed a minute earlier matched + * instantly on a fresh wait. Callers must use a marker unique to the call + * (`echo DONE_$RANDOM`), not a generic one like `BUILD OK`. + * + * @throws {WaitCapacityError} when a waiter cap is already reached. + */ + waitForOutput(sessionId: string, options: OutputWaitOptions): Promise { + const nocase = options.nocase === true; + const needle = options.match; + const needleCmp = nocase ? needle.toLowerCase() : needle; + const base = { + matched: false, + timedOut: false, + immediate: false, + snippet: null, + waitedMs: 0, + timeoutMs: options.timeoutMs, + }; + + if (options.abortSignal?.aborted) return Promise.resolve({ ...base, ended: true, aborted: true }); + if (this.stopped) return Promise.resolve({ ...base, ended: true, aborted: false }); + + // Capacity FIRST, deliberately: the `from=buffer` scan below strips ANSI over (and + // for nocase lowercases again) up to MAX_BUFFER_SCAN_BYTES, and the route has + // already joined the session's whole terminal buffer to produce it. Doing that for + // a request that is about to be refused is unbounded work at request rate with the + // caps providing no backpressure at all. + this.assertCapacity(sessionId, options.owner); + + // `from=buffer`: scan what is already there before blocking. + let carry = ''; + if (options.initialText) { + const text = normalizeForMatch(options.initialText); + const hit = findMatch(text, needleCmp, nocase); + if (hit) { + return Promise.resolve({ + matched: true, + timedOut: false, + immediate: true, + ended: false, + aborted: false, + snippet: hit, + waitedMs: 0, + timeoutMs: options.timeoutMs, + }); + } + carry = tailFor(text, needleCmp.length); + } + + return new Promise((resolve) => { + let set = this.outputWaiters.get(sessionId); + if (!set) { + set = new Set(); + this.outputWaiters.set(sessionId, set); + } + + const waiter: OutputWaiter = { + needle, + needleCmp, + nocase, + carry, + startedAt: Date.now(), + timeoutMs: options.timeoutMs, + owner: options.owner, + timer: setTimeout(() => { + this.removeOutputWaiter(sessionId, waiter); + resolve({ + matched: false, + timedOut: true, + immediate: false, + ended: false, + aborted: false, + snippet: null, + waitedMs: Date.now() - waiter.startedAt, + timeoutMs: waiter.timeoutMs, + }); + }, options.timeoutMs), + settle: resolve, + }; + set.add(waiter); + this.chargeOwner(options.owner); + waiter.detachAbort = this.attachAbort(options.abortSignal, () => { + this.removeOutputWaiter(sessionId, waiter); + resolve({ + matched: false, + timedOut: false, + immediate: false, + ended: true, + aborted: true, + snippet: null, + waitedMs: Date.now() - waiter.startedAt, + timeoutMs: waiter.timeoutMs, + }); + }); + }); + } + + /** + * Feed a raw PTY chunk to this session's output waiters. + * + * Called from the always-attached `terminal` listener, so it sits on the PTY hot + * path: it must stay a single Map lookup when nobody is waiting, which is why the + * no-waiter check comes before the ANSI strip. + * + * ⚠️ A PTY read boundary can fall INSIDE an escape sequence, which is routine under + * tmux (it emits SGR runs constantly). `stripAnsi` needs a complete sequence, so a + * split one used to survive the strip, land in the carry, and split the needle: the + * same text matched or not depending on where the kernel happened to cut the read. + * The incomplete tail is therefore held back in `pendingAnsi` and stripped together + * with the next chunk. That is the same class of bug the carry buffer exists to + * solve, one level down. + * + * @returns how many waiters matched. + */ + notifyOutput(sessionId: string, chunk: string): number { + const set = this.outputWaiters.get(sessionId); + if (!set || set.size === 0) return 0; + + const pending = this.pendingAnsi.get(sessionId); + const split = splitTrailingEscape(pending ? pending + chunk : chunk); + if (split.pending) this.pendingAnsi.set(sessionId, split.pending); + else if (pending !== undefined) this.pendingAnsi.delete(sessionId); + + const text = normalizeForMatch(split.text); + if (!text) return 0; + + let resolved = 0; + for (const waiter of [...set]) { + // Prepend the tail of what this waiter already scanned so a match spanning + // two chunks is still found. Re-scanning the carry cannot double-fire: a + // needle wholly inside the carry would have matched on the previous pass. + const hay = waiter.carry + text; + const hit = findMatch(hay, waiter.needleCmp, waiter.nocase); + if (hit) { + this.removeOutputWaiter(sessionId, waiter); + waiter.settle({ + matched: true, + timedOut: false, + immediate: false, + ended: false, + aborted: false, + snippet: hit, + waitedMs: Date.now() - waiter.startedAt, + timeoutMs: waiter.timeoutMs, + }); + resolved++; + continue; + } + waiter.carry = tailFor(hay, waiter.needleCmp.length); + } + return resolved; + } + + private removeOutputWaiter(sessionId: string, waiter: OutputWaiter): void { + clearTimeout(waiter.timer); + waiter.detachAbort?.(); + const set = this.outputWaiters.get(sessionId); + if (!set) return; + if (!set.delete(waiter)) return; + this.releaseOwner(waiter.owner); + if (set.size === 0) { + this.outputWaiters.delete(sessionId); + // The held-back escape tail belongs to the waiter set; with nobody watching it + // would be a per-session string kept alive for the life of the process. + this.pendingAnsi.delete(sessionId); + } + } + + // ── Teardown ── + + /** + * Resolve every waiter for a session with `ended: true`. + * + * Call on session deletion and PTY teardown. A session that EXITS should + * `notifySignal(id, 'exit')` first, so an `until=exit` caller sees the signal + * rather than a bare `ended`. + * + * @returns how many waiters were resolved. + */ + cancelAll(sessionId: string): number { + let resolved = 0; + + const signals = this.signalWaiters.get(sessionId); + if (signals) { + for (const waiter of [...signals]) { + this.removeSignalWaiter(sessionId, waiter); + waiter.settle({ + signal: null, + timedOut: false, + immediate: false, + ended: true, + aborted: false, + waitedMs: Date.now() - waiter.startedAt, + timeoutMs: waiter.timeoutMs, + }); + resolved++; + } + } + + const outputs = this.outputWaiters.get(sessionId); + if (outputs) { + for (const waiter of [...outputs]) { + this.removeOutputWaiter(sessionId, waiter); + waiter.settle({ + matched: false, + timedOut: false, + immediate: false, + ended: true, + aborted: false, + snippet: null, + waitedMs: Date.now() - waiter.startedAt, + timeoutMs: waiter.timeoutMs, + }); + resolved++; + } + } + + this.pendingAnsi.delete(sessionId); + return resolved; + } + + /** + * Resolve every waiter in the process and refuse every later one. + * + * Shutdown must call this: waiter timers are not unref'd, so a pending 10-minute + * wait would otherwise hold the event loop open and strand its HTTP response. + * + * ⚠️ The latch is the point, not just the sweep. `server.stop()` cancels waiters + * well before `app.close()`, with several awaits in between (orchestrator teardown, + * scheduled-run stops that each await a session stop and a tmux kill), and the + * listener is still accepting requests throughout. A `GET .../wait?timeout=600000` + * landing in that window used to register a fresh waiter that nothing would ever + * cancel, stalling shutdown for up to `MAX_WAIT_MS` — exactly the outcome cancelling + * was meant to prevent. Once latched, a wait resolves at once with `ended: true`. + */ + stop(): number { + this.stopped = true; + return this.cancelEverything(); + } + + /** + * Resolve every waiter in the process. Latches the stopped flag, so this is + * shutdown-shaped; `stop()` is the same call under the name that says so. + */ + cancelEverything(): number { + this.stopped = true; + let resolved = 0; + const ids = new Set([...this.signalWaiters.keys(), ...this.outputWaiters.keys()]); + for (const id of ids) resolved += this.cancelAll(id); + return resolved; + } +} + +// ─── Shared helpers ────────────────────────────────────────────────────────── + +/** + * Escape sequences `stripAnsi` leaves behind, removed so the text this module MATCHES + * against is what the pane visually shows. + * + * `ANSI_ESCAPE_PATTERN_FULL` covers three families — `ESC [ … letter`, + * `ESC ] … (BEL|ST)`, and `ESC =` / `ESC >` — and nothing else. The gap is not + * theoretical: a stock bash prompt emits `arkon@tnode\x1b(B\x1b[m:`, the CSI is removed + * and the charset-select `ESC ( B` is not, so `match=tnode:` **fails on a prompt that + * plainly reads `tnode:`** while `match=(B` succeeds. That is the single most likely + * thing an orchestrating agent tries, and the failure is silent (a full-length timeout). + * + * Two branches, in order: + * - `ESC P|X|^|_ … (BEL|ST)` — DCS / SOS / PM / APC string sequences. The terminator is + * REQUIRED, exactly as `stripAnsi`'s OSC branch requires one: matching an unterminated + * string type with a greedy run would delete every byte after it to the end of the + * window, which is real output an agent is waiting for. + * - `ESC <0x20-0x2f>* <0x30-0x7e>` — the ECMA-48 escape-sequence grammar (zero or more + * intermediate bytes then a final byte). This is what covers `ESC ( B`, `ESC ) 0`, + * `ESC c`, `ESC 7` / `ESC 8`, `ESC # 8` and a stray `ESC \`. + * + * Linear-time by the same argument that holds for `stripAnsi`: every starred class is + * disjoint from what follows it, and `ESC` is excluded from all of them, so two + * candidate runs can never overlap and there is no backtracking to exploit. + * + * ⚠️ This duplicates ANSI knowledge that would be better held once in + * `src/utils/regex-patterns.ts`. It lives here deliberately: `stripAnsi` has many + * consumers (respawn and usage-limit pattern matching, the terminal buffer, search) and + * widening it changes all of them at once, which is not a change to make as a side + * effect of fixing this endpoint. See the follow-up note in + * `tmp/agent-wait-review/report-fix-registry.md`. + */ +// eslint-disable-next-line no-control-regex -- matching raw terminal control bytes is the point +const ANSI_ESCAPE_RESIDUE = /\x1b(?:[PX^_][^\x07\x1b]*(?:\x07|\x1b\\)|[\x20-\x2f]*[\x30-\x7e])/g; + +/** + * The text this module matches against: ANSI-stripped, then residue-stripped. + * + * Everything downstream (the carry, the haystack, the snippet window) is derived from + * the output of this function, so there is exactly one definition of "the matched + * stream" and `findMatch` never sees an escape byte. + */ +function normalizeForMatch(raw: string): string { + return stripAnsi(raw).replace(ANSI_ESCAPE_RESIDUE, ''); +} + +/** + * C0 controls, DEL and the C1 block, minus tab / newline / carriage return. Removed + * from the snippet so no byte that can reprogram the reading agent's terminal, or + * begin a sequence that does, survives into the JSON response. + * + * Still needed after `normalizeForMatch`, which removes escape SEQUENCES: a bare BEL, + * NUL or backspace carries no ESC and reaches the haystack intact. + */ +// eslint-disable-next-line no-control-regex -- removing raw terminal control bytes is the point +const SNIPPET_CONTROL_BYTES = /[\x00-\x08\x0b\x0c\x0e-\x1f\x7f-\x9f]/g; + +/** + * Find `needleCmp` in `hay` and return a bounded window of surrounding text, + * or null when absent. The snippet comes from the ORIGINAL-case text even in + * nocase mode, so callers see what the terminal actually printed. + * + * ## Two texts, and they are not the same string + * + * Precision here matters, because an agent that cannot match something reaches for the + * snippet to work out why: + * + * - **Matched**: `hay`, which is `normalizeForMatch()` output — ANSI-stripped and + * residue-stripped, with nothing else touched. `matched` is decided against exactly + * this, so it is byte-for-byte what a needle must appear in. + * - **Returned**: a bounded window of that same text, then RENDERED for a human or an + * LLM to read: leftover control bytes removed, blank runs collapsed, trimmed. + * + * So the snippet is a rendering of the matched window, not a quotation of it. The + * remaining differences cannot change whether a printable needle matched, but they are + * real: a needle containing `\n\n` or a raw control byte is matchable and would not + * appear verbatim in the snippet. + * + * Blank runs are collapsed for READABILITY ONLY. A real pane pads with dozens of `\r\n` + * between the prompt and the match, which would otherwise fill the whole context window + * with nothing and make the snippet useless to the agent reading it. + * + * Control bytes are dropped because, unlike `GET .../terminal`, this field is documented + * as stripped and its consumer is an agent piping it through `jq` into its OWN terminal + * pane, so a worker printing attacker-influenced bytes could otherwise reset or garble + * the orchestrator's display. Whitespace is preserved; every other C0/C1 control and DEL + * is removed. + */ +function findMatch(hay: string, needleCmp: string, nocase: boolean): string | null { + const cmp = nocase ? hay.toLowerCase() : hay; + const idx = cmp.indexOf(needleCmp); + if (idx === -1) return null; + + // ⚠️ `idx` indexes `cmp`, and the snippet is sliced out of `hay`. Those agree only + // while lowercasing preserves length, which it usually but not always does: + // 'İ' (U+0130) lowercases to TWO code units, so terminal output containing Turkish + // text, a filename or a git author line shifts every later index and the window + // slides off the match entirely. A length compare detects it for the cost of one + // integer read, so the common path stays exactly one `indexOf` and no allocation. + const drift = cmp.length - hay.length; + const start = drift === 0 ? idx : originalIndex(hay, idx, drift); + const matchEnd = drift === 0 ? idx + needleCmp.length : originalIndex(hay, idx + needleCmp.length, drift); + + return hay + .slice(Math.max(0, start - MAX_SNIPPET_CONTEXT), Math.min(hay.length, matchEnd + MAX_SNIPPET_CONTEXT)) + .replace(SNIPPET_CONTROL_BYTES, '') + .replace(/[\r\n]{2,}/g, '\n') + .trim(); +} + +/** + * Map an index in `hay.toLowerCase()` back to the equivalent index in `hay`, for the + * rare case where the two differ in length by `drift`. + * + * `f(i) = hay.slice(0, i).toLowerCase().length` is non-decreasing and satisfies + * `i <= f(i) <= i + drift` (no lowercase mapping ever shortens a string), which bounds + * the answer to `[cmpIdx - drift, cmpIdx]` and lets a binary search find it in + * `log2(drift)` steps rather than walking the string. Ties break low, so an index that + * lands mid-expansion resolves to the character that expanded, which keeps the window + * around the match rather than past it. + */ +function originalIndex(hay: string, cmpIdx: number, drift: number): number { + let lo = Math.max(0, cmpIdx - drift); + let hi = Math.min(hay.length, cmpIdx); + while (lo < hi) { + const mid = Math.ceil((lo + hi) / 2); + if (hay.slice(0, mid).toLowerCase().length <= cmpIdx) lo = mid; + else hi = mid - 1; + } + return lo; +} + +/** + * Tail of `text` to carry into the next chunk. Long enough to complete a straddling + * match and to give a later snippet some leading context. + * + * Callers pass the COMPARISON needle's length, not the original's: under `nocase` the + * text that matches can be longer than the needle as typed (a needle of `İ` compares + * as two code units, so two code units of pane output can satisfy one needle + * character), and a carry sized from the shorter one drops a straddling match. + */ +function tailFor(text: string, needleLength: number): string { + const keep = Math.max(needleLength - 1, MAX_SNIPPET_CONTEXT); + return keep >= text.length ? text : text.slice(text.length - keep); +} + +/** + * Longest tail held back as a possibly-incomplete escape sequence. + * + * Generous enough for a real OSC title (`ESC ] 0 ; BEL`) to survive a chunk + * split, small enough that an unterminated sequence cannot withhold meaningful output: + * past this the tail is treated as ordinary text and released, so a malformed stream + * degrades to the old behavior instead of stalling every match on the session. + */ +const MAX_PENDING_ESCAPE = 512; + +/** + * A tail that is a strict PREFIX of a sequence `normalizeForMatch` would remove: a lone + * ESC, an unterminated `ESC [ …`, an unterminated string type (`ESC ] P X ^ _ …`), or + * intermediate bytes still waiting for their final byte (`ESC (` before its `B`). + * Anything else is either complete or not an escape at all, and both are safe to strip + * now. + * + * ⚠️ The third branch must stay in lockstep with `ANSI_ESCAPE_RESIDUE`'s second one. + * When the residue pass learned about `ESC ( B`, a chunk cut between the `(` and the `B` + * became a NEW way to smuggle an escape into the haystack, since neither half is a + * complete sequence on its own — the same bug one level down, which is what this + * function exists to prevent. + * + * Deliberately without the `g` flag: a global pattern would carry `lastIndex` between + * calls (see the repo-wide global-regex hazard) and this runs once per PTY chunk. + */ +// eslint-disable-next-line no-control-regex -- matching raw terminal control bytes is the point +const INCOMPLETE_ANSI_TAIL = /^\x1b(?:\[[0-9;?]*|[\]PX^_][^\x07\x1b]*|[\x20-\x2f]*)$/; + +/** + * Split a raw chunk into the part that is safe to ANSI-strip now and a trailing + * fragment to hold for the next chunk. + * + * Only the LAST `ESC` can be incomplete: anything before it is followed by an escape + * introducer, so `stripAnsi` will consume or reject it on this pass either way. + */ +function splitTrailingEscape(raw: string): { text: string; pending: string } { + const esc = raw.lastIndexOf('\x1b'); + if (esc === -1) return { text: raw, pending: '' }; + + const tail = raw.slice(esc); + if (tail.length > MAX_PENDING_ESCAPE || !INCOMPLETE_ANSI_TAIL.test(tail)) { + return { text: raw, pending: '' }; + } + return { text: raw.slice(0, esc), pending: tail }; +} + +/** + * Process-wide registry used by the routes and the session/hook wiring. + * Tests should construct their own `SessionWaitRegistry` instead of touching this. + */ +export const sessionWaits = new SessionWaitRegistry(); diff --git a/test/hook-secret-selfheal.test.ts b/test/hook-secret-selfheal.test.ts index 9af794be..7890cb31 100644 --- a/test/hook-secret-selfheal.test.ts +++ b/test/hook-secret-selfheal.test.ts @@ -133,6 +133,25 @@ describe('refreshStaleCodemanHooks', () => { expect(after.hooks.CustomEvent).toEqual(customEvent); }); + // A case can be current on the secret AND the background-wake hook and still carry + // the `-k`-less curl shape, which exits 60 against a self-signed HTTPS API and is + // swallowed by `|| true` — every hook event dead, silently. The refresh must treat + // that as a third stale shape. + it('heals a current-looking block whose hook curls lack -k (HTTPS self-signed installs)', async () => { + const { generateHooksConfig } = await import('../src/hooks-config.js'); + const flagless = JSON.parse(JSON.stringify(generateHooksConfig()).replaceAll('curl -sk ', 'curl -s ')); + writeFileSync(settingsPath, JSON.stringify({ hooks: flagless.hooks }, null, 2)); + + await refreshStaleCodemanHooks(dir); + + const after = readFileSync(settingsPath, 'utf-8'); + expect(after).toContain('curl -sk -X POST'); + expect(after).not.toContain('curl -s -X POST'); + // and the pass is convergent: a second refresh must not rewrite + await refreshStaleCodemanHooks(dir); + expect(readFileSync(settingsPath, 'utf-8')).toBe(after); + }); + it('is a no-op when settings.local.json is absent (does not create one)', async () => { await refreshStaleCodemanHooks(dir); expect(existsSync(settingsPath)).toBe(false); diff --git a/test/hooks-config.test.ts b/test/hooks-config.test.ts index e4cbd254..149ffabe 100644 --- a/test/hooks-config.test.ts +++ b/test/hooks-config.test.ts @@ -95,6 +95,17 @@ describe('generateHooksConfig', () => { expect(notifHooks[0].hooks[0].command).toContain('|| true'); }); + // On --https/tailscale installs CODEMAN_API_URL is HTTPS with a self-signed cert. + // A `-k`-less hook curl exits 60 there, the `|| true` swallows it, and every hook + // event (stop, permission_prompt, elicitation_dialog, idle_prompt, teammate_idle, + // task_completed) dies silently — killing respawn's idle signals and the wait + // endpoints' stop/blocked. The statusline exporter always carried -k; the hooks must too. + it('every hook curl tolerates a self-signed HTTPS API (curl -sk)', () => { + const serialized = JSON.stringify(generateHooksConfig()); + expect(serialized).toContain('curl -sk -X POST'); + expect(serialized).not.toContain('curl -s -X POST'); + }); + it('should set timeout to 10 seconds (hook timeout fields are seconds)', () => { const config = generateHooksConfig(); const notifHooks = config.hooks.Notification as Array<{ hooks: Array<{ timeout: number }> }>; diff --git a/test/http-contract.test.ts b/test/http-contract.test.ts index 74d44e4d..30e46c35 100644 --- a/test/http-contract.test.ts +++ b/test/http-contract.test.ts @@ -89,4 +89,97 @@ describe('Stable HTTP contract (live server)', () => { expect(body.success).toBe(false); expect(body.errorCode).toBe('INVALID_INPUT'); }); + + /** + * The agent wait primitives, through the REAL pipeline. + * + * Their own route tests hand-roll a partial copy of the preSerialization hook that + * maps errorCode to status but does NOT wrap bare payloads — so nothing there + * proves these routes emit a correct envelope, a correct status, or work through + * the /api/v1 alias, and one assertion in them pins `{}` for a response no client + * will ever receive. This is the file whose docstring already claims that scope. + */ + describe('agent wait primitives', () => { + let sessionId: string; + + beforeAll(async () => { + const res = await fetch(`${base}/api/sessions`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({}), + }); + sessionId = (await res.json()).data.session.id; + expect(sessionId).toBeDefined(); + }); + + afterAll(async () => { + await fetch(`${base}/api/sessions/${sessionId}`, { method: 'DELETE' }); + }); + + it('answers a wait timeout as a 200 inside the envelope, on the /api/v1 alias', async () => { + // A timeout is the long-poll SUCCEEDING at "did this happen within N ms?"; a + // 4xx/5xx here would make every poll boundary indistinguishable from a failure. + const res = await fetch(`${base}/api/v1/sessions/${sessionId}/wait?until=working&timeout=1000`); + expect(res.status).toBe(200); + expect(res.headers.get('cache-control')).toBe('no-store'); + + const body = await res.json(); + expect(body.success).toBe(true); + expect(body.data.sessionId).toBe(sessionId); + // The one shape all three wait endpoints share. + expect(body.data.wait.timedOut).toBe(true); + expect(body.data.wait.signal).toBeNull(); + expect(body.data.wait.timeoutMs).toBe(1000); + expect(body.data.wait.until).toEqual(['working']); + }); + + it('returns a contract-shaped 400 for an unknown until token', async () => { + const res = await fetch(`${base}/api/v1/sessions/${sessionId}/wait?until=stpo`); + expect(res.status).toBe(400); + const body = await res.json(); + expect(body.success).toBe(false); + expect(body.errorCode).toBe('INVALID_INPUT'); + expect(body.error).toContain('stpo'); + }); + + it('returns a contract-shaped 400 naming the bad query parameter', async () => { + const res = await fetch(`${base}/api/v1/sessions/${sessionId}/wait?timeout=30s`); + expect(res.status).toBe(400); + const body = await res.json(); + expect(body.errorCode).toBe('INVALID_INPUT'); + expect(body.error).toContain('timeout'); + }); + + it('wraps the non-wait input response as { success: true, data: {} }', async () => { + // What a client actually receives on the fire-and-forget path — NOT the bare + // `{}` the handler returns and the route tests assert. + const res = await fetch(`${base}/api/v1/sessions/${sessionId}/input`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ input: 'hello' }), + }); + expect(res.status).toBe(200); + expect(await res.json()).toEqual({ success: true, data: {} }); + }); + + it('serves wait-output through the same envelope', async () => { + const res = await fetch(`${base}/api/v1/sessions/${sessionId}/wait-output?match=NEVER_APPEARS&timeout=1000`); + expect(res.status).toBe(200); + const body = await res.json(); + expect(body.success).toBe(true); + expect(body.data.wait.matched).toBe(false); + expect(body.data.wait.timedOut).toBe(true); + expect(body.data.wait.match).toBe('NEVER_APPEARS'); + }); + + it('404s an unknown session on both new routes, with the error envelope', async () => { + for (const path of ['wait?until=idle', 'wait-output?match=x']) { + const res = await fetch(`${base}/api/v1/sessions/nonexistent/${path}`); + expect(res.status).toBe(404); + const body = await res.json(); + expect(body.success).toBe(false); + expect(body.errorCode).toBe('NOT_FOUND'); + } + }); + }); }); diff --git a/test/mocks/mock-session.ts b/test/mocks/mock-session.ts index 7d7a5be7..18990a2e 100644 --- a/test/mocks/mock-session.ts +++ b/test/mocks/mock-session.ts @@ -4,6 +4,7 @@ */ import { EventEmitter } from 'node:events'; import { vi } from 'vitest'; +import type { SessionStatus } from '../../src/types.js'; /** * Enhanced mock session for testing RespawnController. @@ -12,8 +13,15 @@ import { vi } from 'vitest'; export class MockSession extends EventEmitter { id: string; workingDir: string = '/tmp/test-workdir'; - status: 'idle' | 'working' = 'idle'; - pid: number = 12345; + /** + * The REAL union, deliberately. This used to be `'idle' | 'working'`, and + * `'working'` is not a `SessionStatus` at all — so `signalForStatus()` fell to its + * `default: null` branch in every route test and the busy / stopped / error halves + * of the immediate-resolve mapping had zero coverage while appearing to be tested. + */ + status: SessionStatus = 'idle'; + /** `null` once the PTY is gone (or before it has ever started) — see `pid` in Session. */ + pid: number | null = 12345; isWorking: boolean = false; private _activeChildProcesses: { pid: number; command: string }[] = []; ralphTracker: null = null; @@ -113,7 +121,9 @@ export class MockSession extends EventEmitter { /** Simulate working state with spinner */ simulateWorking(text: string = 'Thinking'): void { this.simulateTerminalOutput(`${text}... \u280b`); - this.status = 'working'; + // 'busy' is what the real Session sets while a turn is in flight; the old + // 'working' here was the event name, not a status value. + this.status = 'busy'; this.emit('working'); } @@ -182,6 +192,14 @@ export class MockSession extends EventEmitter { return this._muxName; } + /** + * Mirrors `Session.usesMux`. True by default because that is the normal + * configuration, and it is what makes a route's pane-liveness probe reachable: + * `session.pid` is the tmux ATTACH CLIENT, so a mux-backed session's worker can be + * dead while `pid` is still a live number. + */ + usesMux: boolean = true; + /** Check for active child processes (mock returns configurable list) */ getActiveChildProcesses(): { pid: number; command: string }[] { return this._activeChildProcesses; diff --git a/test/routes/session-input-wait.test.ts b/test/routes/session-input-wait.test.ts new file mode 100644 index 00000000..9d1bf965 --- /dev/null +++ b/test/routes/session-input-wait.test.ts @@ -0,0 +1,671 @@ +/** + * @fileoverview Route tests for the `wait` field on `POST /api/sessions/:id/input`. + * + * This endpoint exists to close a race a caller cannot close from outside: between + * the write landing and the session flipping to `working`, a SEPARATE wait sees the + * session still idle and returns instantly, reporting the previous turn as this one. + * Registering the waiter before the write is the whole point, so that is what the + * first test pins. + * + * It also pins that the historical fire-and-forget path is untouched when `wait` is + * absent, that a capacity rejection gives the dedup seq back (otherwise the caller's + * retry is refused as a duplicate and the input is lost by the very mechanism + * reliable delivery exists for), and that `delivered` reports what actually happened + * to the write rather than merely "this was not a duplicate". + * + * Plan: docs/agent-control-plan.md + */ +import { describe, it, expect, afterEach, beforeAll, afterAll, vi } from 'vitest'; +import fastifyCookie from '@fastify/cookie'; +import Fastify, { type FastifyInstance } from 'fastify'; +import type { IncomingMessage } from 'node:http'; +import { registerSessionRoutes, _resetPaneLivenessState } from '../../src/web/routes/session-routes.js'; +import { installRouteErrorHandler } from '../../src/web/route-error-handler.js'; +import { ApiErrorCode, httpStatusForErrorCode } from '../../src/types.js'; +import { createMockRouteContext, type MockRouteContext } from '../mocks/index.js'; +import { sessionWaits } from '../../src/web/session-wait-registry.js'; +import { MAX_WAIT_MS } from '../../src/config/agent-wait.js'; + +// Distinct per file on purpose: the three wait suites share the process-wide +// `sessionWaits` singleton, so a common id let one file's leftover waiter be counted +// by another's assertion. Failed only in a 5-file run, which is how CI runs them. +const SESSION_ID = 'input-wait-session'; +const URL = `/api/sessions/${SESSION_ID}/input`; + +afterEach(() => { + // Deliberately not `cancelEverything()`: it latches the registry's stopped flag, + // which would leave every later test in this file talking to a dead registry. + sessionWaits.cancelAll(SESSION_ID); + _resetPaneLivenessState(); +}); + +async function harness(): Promise<{ app: FastifyInstance; ctx: MockRouteContext; rawRequests: IncomingMessage[] }> { + const app = Fastify({ logger: false }); + await app.register(fastifyCookie); + const ctx = createMockRouteContext({ sessionId: SESSION_ID }); + const rawRequests: IncomingMessage[] = []; + app.addHook('onRequest', async (req) => { + rawRequests.push(req.raw); + }); + + registerSessionRoutes(app, ctx as never); + + app.addHook('preSerialization', (req, reply, payload: unknown, done) => { + const p = payload as { success?: unknown; errorCode?: unknown } | null; + if (p && typeof p === 'object' && p.success === false && reply.statusCode === 200) { + if (typeof p.errorCode === 'string') reply.code(httpStatusForErrorCode(p.errorCode as ApiErrorCode)); + } + return done(null, payload); + }); + + installRouteErrorHandler(app); + await app.ready(); + return { app, ctx, rawRequests }; +} + +const send = (app: FastifyInstance, payload: Record) => + app.inject({ method: 'POST', url: URL, payload }); + +describe('POST /api/sessions/:id/input without wait (unchanged behavior)', () => { + it('returns the historical bare body and registers no waiter', async () => { + const { app } = await harness(); + const res = await send(app, { input: 'hello', useMux: true }); + + expect(res.statusCode).toBe(200); + expect(res.json()).toEqual({}); + expect(sessionWaits.totalWaiterCount()).toBe(0); + }); + + it('still returns before the mux write settles', async () => { + // The fire-and-forget property is why the response is fast; send-and-wait must + // not have turned every input into an awaited tmux round-trip. + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + let resolveWrite: (ok: boolean) => void = () => {}; + session.writeViaMux = () => new Promise((resolve) => (resolveWrite = resolve)); + + const res = await send(app, { input: 'hello', useMux: true }); + expect(res.json()).toEqual({}); + resolveWrite(true); + }); + + it('a tagged duplicate still returns the bare body', async () => { + const { app } = await harness(); + await send(app, { input: 'first', clientId: 'c1', seq: 1 }); + const replay = await send(app, { input: 'first', clientId: 'c1', seq: 1 }); + + expect(replay.json()).toEqual({}); + expect(sessionWaits.totalWaiterCount()).toBe(0); + }); + + it('a failed direct write is still not an error response', async () => { + // A session can legitimately have no PTY yet, and callers have always been able + // to write to one without a 4xx. Only the `wait` path reports delivery. + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.failWrites = true; + + const res = await send(app, { input: 'x' }); + expect(res.statusCode).toBe(200); + expect(res.json()).toEqual({}); + }); +}); + +describe('POST /api/sessions/:id/input with wait', () => { + it('registers the waiter BEFORE the write, so the pre-existing idle state cannot satisfy it', async () => { + // The mock session is idle. A naive send-then-wait would answer immediately with + // that stale idle; this must block until a real transition. + const { app } = await harness(); + const pending = send(app, { input: 'run the tests', useMux: true, wait: true }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(1); + + sessionWaits.notifySignal(SESSION_ID, 'stop'); + const body = (await pending).json(); + expect(body.success).toBe(true); + expect(body.data.delivered).toBe(true); + expect(body.data.duplicate).toBe(false); + expect(body.data.wait.signal).toBe('stop'); + expect(body.data.wait.immediate).toBe(false); + expect(body.data.wait.timedOut).toBe(false); + }); + + it('uses the same data.wait envelope as the two GET endpoints', async () => { + const { app } = await harness(); + const pending = send(app, { input: 'x', wait: 'stop' }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + sessionWaits.notifySignal(SESSION_ID, 'stop'); + const { data } = (await pending).json(); + + expect(Object.keys(data).sort()).toEqual(['delivered', 'duplicate', 'limitPaused', 'status', 'wait']); + expect(data.status).toBe('idle'); + expect(data.limitPaused).toBe(false); + expect(data.wait.aborted).toBe(false); + }); + + it('echoes the effective timeout after clamping', async () => { + // The schema accepts up to 24h; the server caps at MAX_WAIT_MS. Without the echo + // an agent reads the cap as "my 24h wait elapsed" and kills a healthy worker. + const { app } = await harness(); + const pending = send(app, { input: 'x', wait: 'stop', waitTimeout: 86_400_000 }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + sessionWaits.notifySignal(SESSION_ID, 'stop'); + + expect((await pending).json().data.wait.timeoutMs).toBe(MAX_WAIT_MS); + }); + + it('delivers the input before blocking', async () => { + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + const pending = send(app, { input: 'echo hi', useMux: true, wait: 'stop' }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + // The write happened while the request is still open. + expect(session.writeBuffer.join('')).toContain('echo hi'); + + sessionWaits.notifySignal(SESSION_ID, 'stop'); + await pending; + }); + + it('wait: true uses the default signal set', async () => { + const { app } = await harness(); + const pending = send(app, { input: 'x', wait: true }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + sessionWaits.notifySignal(SESSION_ID, 'idle'); + + expect((await pending).json().data.wait.until).toEqual(['stop', 'idle', 'exit']); + }); + + it('accepts an explicit signal list', async () => { + const { app } = await harness(); + const pending = send(app, { input: 'x', wait: 'exit' }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + // Not one of the requested signals: the wait must not resolve on it. + expect(sessionWaits.notifySignal(SESSION_ID, 'idle')).toBe(0); + sessionWaits.notifySignal(SESSION_ID, 'exit'); + + const body = (await pending).json(); + expect(body.data.wait.signal).toBe('exit'); + expect(body.data.wait.until).toEqual(['exit']); + }); + + it('accepts an array, the same grammar the query parameter takes', async () => { + const { app } = await harness(); + const pending = send(app, { input: 'x', wait: ['stop', 'exit'] }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + sessionWaits.notifySignal(SESSION_ID, 'exit'); + + const body = (await pending).json(); + expect(body.data.wait.until).toEqual(['stop', 'exit']); + expect(body.data.wait.signal).toBe('exit'); + }); + + it('rejects an unknown wait signal without writing', async () => { + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + const before = session.writeBuffer.length; + + const res = await send(app, { input: 'x', wait: 'stpo' }); + expect(res.statusCode).toBe(400); + expect(res.json().errorCode).toBe('INVALID_INPUT'); + expect(session.writeBuffer.length).toBe(before); + }); + + it('rejects a hook-only signal for external CLI modes', async () => { + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.mode = 'codex'; + + const res = await send(app, { input: 'x', wait: 'stop' }); + expect(res.statusCode).toBe(400); + expect(res.json().error).toContain('codex'); + }); + + it('rejects a hook-only signal for a shell session too', async () => { + // Not an external CLI, but a plain bash PTY installs no hooks either, so `stop` + // could only ever time out. + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.mode = 'shell'; + + const res = await send(app, { input: 'x', wait: 'stop' }); + expect(res.statusCode).toBe(400); + expect(res.json().error).toContain('shell'); + }); + + it('times out as a 200, like the standalone wait', async () => { + const { app } = await harness(); + const res = await send(app, { input: 'x', wait: 'blocked', waitTimeout: 1 }); + + expect(res.statusCode).toBe(200); + const body = res.json(); + expect(body.data.delivered).toBe(true); + expect(body.data.wait.timedOut).toBe(true); + expect(body.data.wait.signal).toBeNull(); + }); + + it('resolves with ended when the session goes away mid-wait', async () => { + const { app } = await harness(); + const pending = send(app, { input: 'x', wait: 'stop' }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + sessionWaits.cancelAll(SESSION_ID); + + expect((await pending).json().data.wait.ended).toBe(true); + }); + + it('a request-body close does NOT abort the wait', async () => { + // The regression this pins: on a POST the request stream closes as soon as the + // body has been read, well before the handler blocks. Treating that as a hang-up + // aborted every send-and-wait instantly. Hang-up handling itself is proven over + // real HTTP at the bottom of this file, because inject cannot produce a socket. + const { app, rawRequests } = await harness(); + const pending = send(app, { input: 'x', wait: 'stop', waitTimeout: 600_000 }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(1); + + rawRequests[rawRequests.length - 1].emit('close'); + await new Promise((resolve) => setTimeout(resolve, 20)); + + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(1); + + sessionWaits.notifySignal(SESSION_ID, 'stop'); + const body = (await pending).json(); + expect(body.data.wait.aborted).toBe(false); + expect(body.data.wait.signal).toBe('stop'); + }); + + it('a duplicate skips the write but answers from current state instead of hanging', async () => { + // The original turn is long over, so requiring a fresh transition here would + // block a redelivery until timeout for no reason. + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + await send(app, { input: 'first', clientId: 'c1', seq: 1 }); + const before = session.writeBuffer.length; + + const replay = await send(app, { input: 'first', clientId: 'c1', seq: 1, wait: 'idle' }); + const body = replay.json(); + + expect(body.data.duplicate).toBe(true); + expect(body.data.delivered).toBe(false); + expect(body.data.wait.immediate).toBe(true); + expect(body.data.wait.signal).toBe('idle'); + expect(session.writeBuffer.length).toBe(before); + }); + + it('a duplicate on a BUSY session answers working, not idle', async () => { + // Previously unreachable: MockSession's status was 'working', which is not a + // SessionStatus, so signalForStatus fell through to null and this combination + // silently proved nothing. + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + await send(app, { input: 'first', clientId: 'c2', seq: 1 }); + session.status = 'busy'; + + const replay = await send(app, { input: 'first', clientId: 'c2', seq: 1, wait: 'working' }); + const body = replay.json(); + + expect(body.data.duplicate).toBe(true); + expect(body.data.wait.signal).toBe('working'); + expect(body.data.wait.immediate).toBe(true); + expect(body.data.status).toBe('busy'); + }); + + it('gives the dedup seq back when a full waiter pool rejects the request', async () => { + // Otherwise the caller's retry is refused as a duplicate and the input vanishes. + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + + const pendings = []; + for (let i = 0; i < 16; i++) { + pendings.push(app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop` })); + } + await new Promise((resolve) => setTimeout(resolve, 40)); + expect(sessionWaits.waiterCount(SESSION_ID)).toBe(16); + + const rejected = await send(app, { input: 'x', clientId: 'c9', seq: 7, wait: true }); + expect(rejected.statusCode).toBe(409); + expect(rejected.json().errorCode).toBe('SESSION_BUSY'); + // Nothing written... + expect(session.writeBuffer.join('')).not.toContain('x'); + + sessionWaits.cancelAll(SESSION_ID); + await Promise.all(pendings); + + // ...and the same seq is accepted on retry rather than treated as a replay. + const retry = await send(app, { input: 'x', clientId: 'c9', seq: 7 }); + expect(retry.json()).toEqual({}); + expect(session.writeBuffer.join('')).toContain('x'); + }); + + it('a null wait is treated as absent, not as a validation error', async () => { + // Zod .optional() rejects null, and a third-party caller building the body with + // JSON.stringify keeps an explicit null on the wire. + const { app } = await harness(); + const res = await send(app, { input: 'x', wait: null, waitTimeout: null }); + + expect(res.statusCode).toBe(200); + expect(res.json()).toEqual({}); + }); + + it('wait: false is treated as absent', async () => { + const { app } = await harness(); + const res = await send(app, { input: 'x', wait: false }); + + expect(res.statusCode).toBe(200); + expect(res.json()).toEqual({}); + expect(sessionWaits.totalWaiterCount()).toBe(0); + }); + + it('falls back to a direct write when the mux write fails, and still waits', async () => { + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + session.writeViaMux = async () => false; + + const pending = send(app, { input: 'fallback me', useMux: true, wait: 'stop' }); + await new Promise((resolve) => setTimeout(resolve, 20)); + expect(session.writeBuffer.join('')).toContain('fallback me'); + + sessionWaits.notifySignal(SESSION_ID, 'stop'); + const body = (await pending).json(); + expect(body.data.wait.signal).toBe('stop'); + // The fallback write succeeded, so the input really was delivered. + expect(body.data.delivered).toBe(true); + }); +}); + +describe('POST /api/sessions/:id/input: delivered reports the write, not just the dedup', () => { + it('reports delivered:false when BOTH write paths fail, instead of claiming delivery', async () => { + // A worker whose PTY has exited fails writeViaMux AND write. Reporting + // "delivered, but it timed out" points the agent at waiting longer; the truth is + // "restart the worker". + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + session.failWrites = true; + + const res = await send(app, { input: 'run the tests', useMux: true, wait: 'stop', waitTimeout: 600_000 }); + const body = res.json(); + + expect(res.statusCode).toBe(200); + expect(body.data.delivered).toBe(false); + expect(body.data.duplicate).toBe(false); + }); + + it('does not block for the full timeout on an input it knows never landed', async () => { + // The waiter has to be registered before the write, so it exists by the time the + // failure is known; releasing it immediately is what keeps the caller from + // waiting ten minutes for a turn that cannot start. + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.failWrites = true; + + const started = Date.now(); + const body = (await send(app, { input: 'x', useMux: true, wait: 'stop', waitTimeout: 600_000 })).json(); + + expect(Date.now() - started).toBeLessThan(2_000); + expect(body.data.delivered).toBe(false); + expect(body.data.wait.timedOut).toBe(false); + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(0); + }); + + it('reports delivered:false for a failed direct (non-mux) write too', async () => { + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.failWrites = true; + + const body = (await send(app, { input: 'x', wait: 'stop', waitTimeout: 600_000 })).json(); + expect(body.data.delivered).toBe(false); + }); + + it('rolls the dedup seq back when the write failed, so a retry is not a duplicate', async () => { + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + session.failWrites = true; + + const first = (await send(app, { input: 'x', clientId: 'c3', seq: 4, wait: 'stop', waitTimeout: 600_000 })).json(); + expect(first.data.delivered).toBe(false); + + session.failWrites = false; + const retry = (await send(app, { input: 'x', clientId: 'c3', seq: 4, wait: 'stop', waitTimeout: 1 })).json(); + expect(retry.data.duplicate).toBe(false); + expect(retry.data.delivered).toBe(true); + expect(session.writeBuffer.join('')).toContain('x'); + }); +}); + +/** + * Client-hang-up handling, over REAL HTTP. + * + * `app.inject()` never emits a `close` event at all, so the entire abort path is + * invisible to every other test in this file — and the failure it hides is not + * subtle. On a POST, `req.raw` emits `'close'` as soon as the request BODY finishes + * streaming, which happens before the handler blocks (+1ms, `aborted: false`) and is + * indistinguishable from a real hang-up at +0ms. A request-side abort listener + * therefore cancels every send-and-wait instantly: `POST .../input {wait:"exit", + * waitTimeout:10000}` came back in 23ms with `ended:true, aborted:true, waitedMs:6`, + * i.e. the feature was dead while all 27 inject-based tests above stayed green. + * + * GET survives a request-side listener because it has no body to finish, which is + * exactly why this regression needs a POST and a real socket to catch. + */ +describe('POST /api/sessions/:id/input over real HTTP: hang-up handling', () => { + const PORT = 3181; + const base = `http://127.0.0.1:${PORT}`; + let app: FastifyInstance; + + beforeAll(async () => { + app = (await harness()).app; + await app.listen({ port: PORT, host: '127.0.0.1' }); + }); + + afterAll(async () => { + // fetch keeps its sockets alive, and `app.close()` waits for idle connections, + // so without this the teardown hook times out. + app.server.closeAllConnections(); + await app.close(); + }); + + it('a send-and-wait that is NOT aborted blocks for its full timeout', async () => { + // The regression: this returned in ~20ms with aborted:true. + const started = Date.now(); + const res = await fetch(`${base}/api/sessions/${SESSION_ID}/input`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ input: 'run the tests', wait: 'exit', waitTimeout: 2000 }), + }); + const elapsed = Date.now() - started; + const body = await res.json(); + + expect(body.data.wait.aborted).toBe(false); + expect(body.data.wait.timedOut).toBe(true); + expect(body.data.wait.waitedMs).toBeGreaterThan(1500); + expect(elapsed).toBeGreaterThan(1500); + expect(body.data.delivered).toBe(true); + }); + + it('a send-and-wait aborted mid-flight frees its waiter', async () => { + const controller = new AbortController(); + const pending = fetch(`${base}/api/sessions/${SESSION_ID}/input`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ input: 'x', wait: 'exit', waitTimeout: 600_000 }), + signal: controller.signal, + }).catch(() => 'aborted'); + + await new Promise((resolve) => setTimeout(resolve, 150)); + // Still parked: the body finished streaming long ago, and that must not count. + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(1); + + controller.abort(); + expect(await pending).toBe('aborted'); + await new Promise((resolve) => setTimeout(resolve, 100)); + + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(0); + }); + + it('a GET wait behaves the same way on both counts', async () => { + const notAborted = await fetch(`${base}/api/sessions/${SESSION_ID}/wait?until=stop&timeout=1500`); + const body = await notAborted.json(); + expect(body.data.wait.aborted).toBe(false); + expect(body.data.wait.timedOut).toBe(true); + + const controller = new AbortController(); + const pending = fetch(`${base}/api/sessions/${SESSION_ID}/wait?until=stop&timeout=600000`, { + signal: controller.signal, + }).catch(() => 'aborted'); + await new Promise((resolve) => setTimeout(resolve, 100)); + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(1); + + controller.abort(); + expect(await pending).toBe('aborted'); + await new Promise((resolve) => setTimeout(resolve, 100)); + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(0); + }); + + it('wait-output frees its waiter on hang-up too', async () => { + const controller = new AbortController(); + const pending = fetch(`${base}/api/sessions/${SESSION_ID}/wait-output?match=NEVER&timeout=600000`, { + signal: controller.signal, + }).catch(() => 'aborted'); + await new Promise((resolve) => setTimeout(resolve, 100)); + expect(sessionWaits.outputWaiterCount(SESSION_ID)).toBe(1); + + controller.abort(); + expect(await pending).toBe('aborted'); + await new Promise((resolve) => setTimeout(resolve, 100)); + expect(sessionWaits.outputWaiterCount(SESSION_ID)).toBe(0); + }); +}); + +/** + * Send-and-wait against a tmux worker that has already died. + * + * `tmux send-keys` SUCCEEDS against a dead pane, so `writeViaMux` returns true and the + * old `delivered` was true for bytes written into a corpse — with `timedOut: true` + * alongside it, which tells an agent to wait longer when the truth is "restart the + * worker". Live: `pane_dead=1 status=42`, Codeman `pid=309406 status=idle`, + * `delivered: true`. + */ +describe('POST /api/sessions/:id/input: the pane is dead', () => { + function setPaneDead(ctx: MockRouteContext, dead: boolean) { + (ctx.mux as unknown as { isPaneDead: (n: string) => boolean }).isPaneDead = () => dead; + } + + it('reports delivered:false even though the mux write "succeeded"', async () => { + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + setPaneDead(ctx, true); + + const res = await send(app, { input: 'run the tests', useMux: true, wait: 'stop', waitTimeout: 600_000 }); + const body = res.json(); + + // The write itself did not fail — that is the whole trap. + expect(session.writeBuffer.join('')).toContain('run the tests'); + expect(body.data.delivered).toBe(false); + expect(body.data.duplicate).toBe(false); + }); + + it('returns at once instead of blocking on a turn that cannot start', async () => { + const { app, ctx } = await harness(); + setPaneDead(ctx, true); + + const started = Date.now(); + const body = (await send(app, { input: 'x', useMux: true, wait: 'stop', waitTimeout: 600_000 })).json(); + + expect(Date.now() - started).toBeLessThan(2_000); + expect(body.data.wait.timedOut).toBe(false); + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(0); + }); + + it('rolls the dedup seq back, so a retry against a restarted worker is not a duplicate', async () => { + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + setPaneDead(ctx, true); + + const dead = (await send(app, { input: 'x', useMux: true, clientId: 'c7', seq: 3, wait: 'stop' })).json(); + expect(dead.data.delivered).toBe(false); + + // Worker restarted. + setPaneDead(ctx, false); + _resetPaneLivenessState(); + const retry = ( + await send(app, { input: 'x', useMux: true, clientId: 'c7', seq: 3, wait: 'stop', waitTimeout: 1 }) + ).json(); + expect(retry.data.duplicate).toBe(false); + expect(retry.data.delivered).toBe(true); + expect(session.writeBuffer.join('')).toContain('x'); + }); + + it('a live pane still reports delivered:true', async () => { + const { app, ctx } = await harness(); + setPaneDead(ctx, false); + + const pending = send(app, { input: 'x', useMux: true, wait: 'stop' }); + await new Promise((resolve) => setTimeout(resolve, 20)); + sessionWaits.notifySignal(SESSION_ID, 'stop'); + + expect((await pending).json().data.delivered).toBe(true); + }); + + it('never probes tmux on the plain (non-wait) input path', async () => { + // The browser sends thousands of these per session; they must not exec tmux. + const { app, ctx } = await harness(); + const probe = vi.fn(() => false); + (ctx.mux as unknown as { isPaneDead: (n: string) => boolean }).isPaneDead = probe as never; + + await send(app, { input: 'hello', useMux: true }); + await send(app, { input: 'hello again', useMux: true, clientId: 'c1', seq: 1 }); + + expect(probe).not.toHaveBeenCalled(); + }); +}); + +describe('POST /api/sessions/:id/input: `aborted` stays a client-side fact', () => { + it('reports aborted:false when the SERVER released the waiter after a failed delivery', async () => { + // api-reference guarantees a client never sees `aborted: true`, because it means + // "you hung up, nobody is reading this". The release below is the server giving up + // on a write that failed — and the client IS reading the response, so reporting + // `aborted: true` would both break that guarantee and hand an agent a second, + // contradictory reason for an outcome `delivered: false` already explains. + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.failWrites = true; + + const body = (await send(app, { input: 'x', useMux: true, wait: 'stop', waitTimeout: 600_000 })).json(); + + expect(body.data.delivered).toBe(false); + expect(body.data.wait.aborted).toBe(false); + expect(body.data.wait.ended).toBe(true); + expect(body.data.wait.timedOut).toBe(false); + }); + + it('the same holds for a dead pane', async () => { + const { app, ctx } = await harness(); + (ctx.mux as unknown as { isPaneDead: () => boolean }).isPaneDead = () => true; + + const body = (await send(app, { input: 'x', useMux: true, wait: 'stop', waitTimeout: 600_000 })).json(); + expect(body.data.wait.aborted).toBe(false); + }); +}); + +describe('POST /api/sessions/:id/input: an oversized waitTimeout clamps', () => { + it('accepts a value above the old schema ceiling and reports the clamp', async () => { + const { app } = await harness(); + const pending = send(app, { input: 'x', wait: 'stop', waitTimeout: 99_999_999_999 }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + sessionWaits.notifySignal(SESSION_ID, 'stop'); + + const body = (await pending).json(); + expect(body.data.wait.timeoutMs).toBe(MAX_WAIT_MS); + }); + + it('still rejects a non-integer or negative waitTimeout', async () => { + const { app } = await harness(); + for (const value of [-1, 0, 1.5]) { + const res = await send(app, { input: 'x', wait: 'stop', waitTimeout: value }); + expect(res.statusCode, `waitTimeout=${value}`).toBe(400); + } + }); +}); diff --git a/test/routes/session-wait-output-routes.test.ts b/test/routes/session-wait-output-routes.test.ts new file mode 100644 index 00000000..6f460019 --- /dev/null +++ b/test/routes/session-wait-output-routes.test.ts @@ -0,0 +1,432 @@ +/** + * @fileoverview Route tests for `GET /api/sessions/:id/wait-output`. + * + * Same 200-on-timeout contract and same `data.wait` envelope as `/wait`. The + * additional things pinned here: + * - matching is LITERAL, and a `regex` parameter is rejected rather than ignored, + * so an agent that assumed herdr's `--regex` cannot silently wait on the wrong thing; + * - `from=buffer` scans what already scrolled past, bounded to a tail of the buffer, + * and is charged against the waiter cap BEFORE it materializes that buffer; + * - a chunk-straddling match still resolves, since PTY chunking is arbitrary; + * - a client that hangs up frees its waiter instead of holding it to the timeout. + * + * Plan: docs/agent-control-plan.md + */ +import { describe, it, expect, afterEach, vi } from 'vitest'; +import fastifyCookie from '@fastify/cookie'; +import Fastify, { type FastifyInstance } from 'fastify'; +import type { ServerResponse } from 'node:http'; +import { registerSessionRoutes, _resetPaneLivenessState } from '../../src/web/routes/session-routes.js'; +import { createSessionListeners, attachSessionListeners } from '../../src/web/session-listener-wiring.js'; +import { installRouteErrorHandler } from '../../src/web/route-error-handler.js'; +import { ApiErrorCode, httpStatusForErrorCode } from '../../src/types.js'; +import { createMockRouteContext, type MockRouteContext } from '../mocks/index.js'; +import { sessionWaits } from '../../src/web/session-wait-registry.js'; +import { MAX_MATCH_LENGTH, MAX_BUFFER_SCAN_BYTES, MAX_WAIT_MS } from '../../src/config/agent-wait.js'; + +// Distinct per file on purpose: the three wait suites share the process-wide +// `sessionWaits` singleton, so a common id let one file's leftover waiter be counted +// by another's assertion. Failed only in a 5-file run, which is how CI runs them. +const SESSION_ID = 'wait-output-session'; +const URL = `/api/sessions/${SESSION_ID}/wait-output`; + +afterEach(() => { + // Deliberately not `cancelEverything()`: it latches the registry's stopped flag, + // which would leave every later test in this file talking to a dead registry. + sessionWaits.cancelAll(SESSION_ID); + _resetPaneLivenessState(); +}); + +/** Mirrors production's errorCode-to-status mapping; without it negative cases pass vacuously. */ +async function harness(): Promise<{ app: FastifyInstance; ctx: MockRouteContext; rawReplies: ServerResponse[] }> { + const app = Fastify({ logger: false }); + await app.register(fastifyCookie); + const ctx = createMockRouteContext({ sessionId: SESSION_ID }); + const rawReplies: ServerResponse[] = []; + app.addHook('onRequest', async (req, reply) => { + // The RESPONSE, because that is what the handler's hang-up detection listens to. + rawReplies.push(reply.raw); + }); + + registerSessionRoutes(app, ctx as never); + + app.addHook('preSerialization', (req, reply, payload: unknown, done) => { + const p = payload as { success?: unknown; errorCode?: unknown } | null; + if (p && typeof p === 'object' && p.success === false && reply.statusCode === 200) { + if (typeof p.errorCode === 'string') reply.code(httpStatusForErrorCode(p.errorCode as ApiErrorCode)); + } + return done(null, payload); + }); + + installRouteErrorHandler(app); + await app.ready(); + return { app, ctx, rawReplies }; +} + +describe('GET /api/sessions/:id/wait-output', () => { + it('resolves when the string appears on the stream', async () => { + const { app } = await harness(); + const pending = app.inject({ method: 'GET', url: `${URL}?match=BUILD%20OK` }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + expect(sessionWaits.outputWaiterCount(SESSION_ID)).toBe(1); + sessionWaits.notifyOutput(SESSION_ID, 'running tests...\nBUILD OK\n'); + + const body = (await pending).json(); + expect(body.success).toBe(true); + expect(body.data.wait.matched).toBe(true); + expect(body.data.wait.immediate).toBe(false); + expect(body.data.wait.snippet).toContain('BUILD OK'); + expect(body.data.wait.match).toBe('BUILD OK'); + expect(body.data.sessionId).toBe(SESSION_ID); + }); + + it('uses the same data.wait envelope as /wait, so one client helper reads both', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `${URL}?match=never&timeout=1` }); + + const { data } = res.json(); + expect(Object.keys(data).sort()).toEqual(['limitPaused', 'sessionId', 'status', 'wait']); + expect(data.matched).toBeUndefined(); + expect(data.wait.matched).toBe(false); + expect(data.wait.aborted).toBe(false); + }); + + it('sends Cache-Control: no-store', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `${URL}?match=never&timeout=1` }); + + expect(res.headers['cache-control']).toBe('no-store'); + }); + + it('echoes the effective timeout after clamping', async () => { + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.terminalBuffer = 'BUILD OK\n'; + + const res = await app.inject({ method: 'GET', url: `${URL}?match=BUILD%20OK&from=buffer&timeout=1800000` }); + expect(res.json().data.wait.timeoutMs).toBe(MAX_WAIT_MS); + }); + + it('strips ANSI before matching, so colored output still matches', async () => { + const { app } = await harness(); + const pending = app.inject({ method: 'GET', url: `${URL}?match=BUILD%20OK` }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + sessionWaits.notifyOutput(SESSION_ID, '\x1b[32mBUILD\x1b[0m OK\n'); + + expect((await pending).json().data.wait.matched).toBe(true); + }); + + it('matches across a chunk boundary', async () => { + const { app } = await harness(); + const pending = app.inject({ method: 'GET', url: `${URL}?match=BUILD%20OK` }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + sessionWaits.notifyOutput(SESSION_ID, 'trailing text BUIL'); + sessionWaits.notifyOutput(SESSION_ID, 'D OK done'); + + expect((await pending).json().data.wait.matched).toBe(true); + }); + + it('is case-sensitive by default and honors nocase=1', async () => { + const { app } = await harness(); + + const strict = app.inject({ method: 'GET', url: `${URL}?match=build%20ok&timeout=1` }); + await new Promise((resolve) => setTimeout(resolve, 20)); + sessionWaits.notifyOutput(SESSION_ID, 'BUILD OK'); + expect((await strict).json().data.wait.timedOut).toBe(true); + + const loose = app.inject({ method: 'GET', url: `${URL}?match=build%20ok&nocase=1` }); + await new Promise((resolve) => setTimeout(resolve, 20)); + sessionWaits.notifyOutput(SESSION_ID, 'BUILD OK'); + const body = (await loose).json(); + expect(body.data.wait.matched).toBe(true); + // Reported in the terminal's own casing, not the caller's. + expect(body.data.wait.snippet).toContain('BUILD OK'); + }); + + it('from=buffer resolves immediately against output that already scrolled past', async () => { + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.terminalBuffer = 'earlier output\n\x1b[32mBUILD OK\x1b[0m\n'; + + const res = await app.inject({ method: 'GET', url: `${URL}?match=BUILD%20OK&from=buffer` }); + const body = res.json(); + expect(body.data.wait.matched).toBe(true); + expect(body.data.wait.immediate).toBe(true); + expect(body.data.wait.waitedMs).toBe(0); + expect(sessionWaits.totalWaiterCount()).toBe(0); + }); + + it('from=buffer only scans a bounded tail', async () => { + const { app, ctx } = await harness(); + // Old marker pushed past the scan window by newer output. + ctx.sessions.get(SESSION_ID)!.terminalBuffer = `ANCIENT${'x'.repeat(MAX_BUFFER_SCAN_BYTES + 1000)}`; + + const res = await app.inject({ method: 'GET', url: `${URL}?match=ANCIENT&from=buffer&timeout=1` }); + expect(res.statusCode).toBe(200); + expect(res.json().data.wait.timedOut).toBe(true); + }); + + it('defaults to from=now, ignoring what is already in the buffer', async () => { + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.terminalBuffer = 'BUILD OK happened before you asked\n'; + + const res = await app.inject({ method: 'GET', url: `${URL}?match=BUILD%20OK&timeout=1` }); + expect(res.json().data.wait.timedOut).toBe(true); + }); + + it('answers 200 with timedOut on timeout, never an error status', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `${URL}?match=never&timeout=1` }); + + expect(res.statusCode).toBe(200); + const body = res.json(); + expect(body.success).toBe(true); + expect(body.data.wait.timedOut).toBe(true); + expect(body.data.wait.matched).toBe(false); + expect(body.data.wait.snippet).toBeNull(); + }); + + it('resolves with ended when the session goes away mid-wait', async () => { + const { app } = await harness(); + const pending = app.inject({ method: 'GET', url: `${URL}?match=never` }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + sessionWaits.cancelAll(SESSION_ID); + + const body = (await pending).json(); + expect(body.success).toBe(true); + expect(body.data.wait.ended).toBe(true); + expect(body.data.wait.matched).toBe(false); + }); + + it('frees its waiter when the RESPONSE socket closes early', async () => { + // The injected response is genuinely destroyed by then, so the freed slot is what + // this can assert; the wire-level behaviour is pinned over real HTTP in + // session-input-wait.test.ts. + const { app, rawReplies } = await harness(); + const pending = app.inject({ method: 'GET', url: `${URL}?match=never&timeout=600000` }).then( + () => 'completed', + () => 'destroyed' + ); + + await new Promise((resolve) => setTimeout(resolve, 20)); + expect(sessionWaits.outputWaiterCount(SESSION_ID)).toBe(1); + + rawReplies[rawReplies.length - 1].emit('close'); + await new Promise((resolve) => setTimeout(resolve, 10)); + + expect(sessionWaits.outputWaiterCount(SESSION_ID)).toBe(0); + expect(await pending).toBe('destroyed'); + }); + + it('rejects a regex parameter instead of silently ignoring it', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `${URL}?match=x®ex=%5EBUILD.*OK%24` }); + + expect(res.statusCode).toBe(400); + const body = res.json(); + expect(body.errorCode).toBe('INVALID_INPUT'); + expect(body.error).toContain('match='); + }); + + it('requires match', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: URL }); + + expect(res.statusCode).toBe(400); + expect(res.json().errorCode).toBe('INVALID_INPUT'); + }); + + it('rejects an empty or oversized match, and says which parameter was wrong', async () => { + const { app } = await harness(); + + const empty = await app.inject({ method: 'GET', url: `${URL}?match=` }); + expect(empty.statusCode).toBe(400); + expect(empty.json().error).toContain('match'); + + const huge = await app.inject({ method: 'GET', url: `${URL}?match=${'x'.repeat(MAX_MATCH_LENGTH + 1)}` }); + expect(huge.statusCode).toBe(400); + expect(huge.json().error).toContain('match'); + }); + + it('rejects a non-numeric timeout', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `${URL}?match=x&timeout=soon` }); + + expect(res.statusCode).toBe(400); + expect(res.json().errorCode).toBe('INVALID_INPUT'); + expect(res.json().error).toContain('timeout'); + }); + + it('rejects an unknown from value', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `${URL}?match=x&from=history` }); + + expect(res.statusCode).toBe(400); + }); + + it('404s an unknown session', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: '/api/sessions/nope/wait-output?match=x' }); + + expect(res.statusCode).toBe(404); + expect(res.json().success).toBe(false); + }); + + it('output waiters share the session waiter cap with signal waiters', async () => { + const { app } = await harness(); + const pendings = []; + for (let i = 0; i < 8; i++) { + pendings.push(app.inject({ method: 'GET', url: `${URL}?match=never${i}` })); + } + for (let i = 0; i < 8; i++) { + pendings.push(app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop` })); + } + await new Promise((resolve) => setTimeout(resolve, 40)); + expect(sessionWaits.waiterCount(SESSION_ID)).toBe(16); + + const overflow = await app.inject({ method: 'GET', url: `${URL}?match=one-too-many` }); + expect(overflow.statusCode).toBe(409); + expect(overflow.json().errorCode).toBe('SESSION_BUSY'); + + sessionWaits.cancelAll(SESSION_ID); + await Promise.all(pendings); + }); + + it('checks the cap BEFORE materializing the terminal buffer', async () => { + // `session.terminalBuffer` joins the whole 32MB accumulator. Paying that for a + // request that is about to be refused turns the cap into an amplifier: a caller + // already at the limit can loop `from=buffer` at full speed and never register a + // waiter, so nothing bounds the work. + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + const bufferReads = vi.fn(() => 'nothing to see'); + Object.defineProperty(session, 'terminalBuffer', { get: bufferReads, configurable: true }); + + const pendings = []; + for (let i = 0; i < 16; i++) { + pendings.push(app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop` })); + } + await new Promise((resolve) => setTimeout(resolve, 40)); + expect(sessionWaits.waiterCount(SESSION_ID)).toBe(16); + + const overflow = await app.inject({ method: 'GET', url: `${URL}?match=x&from=buffer` }); + expect(overflow.json().errorCode).toBe('SESSION_BUSY'); + expect(bufferReads).not.toHaveBeenCalled(); + + sessionWaits.cancelAll(SESSION_ID); + await Promise.all(pendings); + }); +}); + +/** + * The LIVE-STREAM half, wired the way production wires it. + * + * Everything above drives `sessionWaits.notifyOutput()` directly, which is the + * registry's API, not the path a real session takes. Deleting the one line that + * connects them — `sessionWaits.notifyOutput(session.id, data)` in the `terminal` + * listener — left all four wait suites green and survived a full `test:ci` sweep, so + * `wait-output?from=now` (the entire live mode, and the one the skill's recipes are + * built on) could ship severed with nothing to show for it. + * + * These go through `createSessionListeners` so the wiring itself is what is pinned. + */ +describe('GET /api/sessions/:id/wait-output: fed by the real terminal listener', () => { + function stubDeps() { + return { + broadcast: vi.fn(), + batchTerminalData: vi.fn(), + batchTaskUpdate: vi.fn(), + broadcastSessionStateDebounced: vi.fn(), + sendPushNotifications: vi.fn(), + persistSessionState: vi.fn(), + getSessionStateWithRespawn: vi.fn(() => ({})), + getRunSummaryTracker: vi.fn(() => undefined), + stopTranscriptWatcher: vi.fn(), + cleanupSessionBatches: vi.fn(), + cancelPersistDebounce: vi.fn(), + removeRunSummaryTracker: vi.fn(), + removeSessionListenerRefs: vi.fn(), + cleanupRespawnOnExit: vi.fn(), + getStore: vi.fn(() => ({ updateRalphState: vi.fn() })), + registerAttachment: vi.fn(async () => {}), + }; + } + + it('matches output emitted by the session, not injected into the registry', async () => { + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + const deps = stubDeps(); + attachSessionListeners(session as never, createSessionListeners(session as never, deps as never)); + + const pending = app.inject({ method: 'GET', url: `${URL}?match=LIVE_STREAM_HIT` }); + await new Promise((resolve) => setTimeout(resolve, 20)); + + // What a PTY chunk actually does: the session emits `terminal`. + session.simulateTerminalOutput('$ echo LIVE_STREAM_HIT\r\nLIVE_STREAM_HIT\r\n'); + + const body = (await pending).json(); + expect(body.data.wait.matched).toBe(true); + expect(body.data.wait.snippet).toContain('LIVE_STREAM_HIT'); + // The listener must still forward to the SSE batcher; the wait feed is additive. + expect(deps.batchTerminalData).toHaveBeenCalled(); + }); + + it('matches across chunk boundaries through the listener', async () => { + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + attachSessionListeners(session as never, createSessionListeners(session as never, stubDeps() as never)); + + const pending = app.inject({ method: 'GET', url: `${URL}?match=SPLIT_MARKER` }); + await new Promise((resolve) => setTimeout(resolve, 20)); + session.simulateTerminalOutput('noise SPLIT_'); + session.simulateTerminalOutput('MARKER more noise'); + + expect((await pending).json().data.wait.matched).toBe(true); + }); + + it('strips ANSI on the way through the listener', async () => { + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + attachSessionListeners(session as never, createSessionListeners(session as never, stubDeps() as never)); + + const pending = app.inject({ method: 'GET', url: `${URL}?match=COLORED%20HIT` }); + await new Promise((resolve) => setTimeout(resolve, 20)); + session.simulateAnsiOutput('COLORED HIT'); + + expect((await pending).json().data.wait.matched).toBe(true); + }); +}); + +describe('GET /api/sessions/:id/wait-output: a dead tmux worker', () => { + it('releases an output waiter when the worker dies while it is parked', async () => { + // The feed simply stops: no exit event, no further chunks, nothing to match. + const { app, ctx } = await harness(); + let dead = false; + (ctx.mux as unknown as { isPaneDead: () => boolean }).isPaneDead = () => dead; + + const pending = app.inject({ method: 'GET', url: `${URL}?match=NEVER&timeout=600000` }); + await new Promise((resolve) => setTimeout(resolve, 50)); + expect(sessionWaits.outputWaiterCount(SESSION_ID)).toBe(1); + + dead = true; + const body = (await pending).json(); + expect(body.data.wait.ended).toBe(true); + expect(body.data.wait.timedOut).toBe(false); + expect(sessionWaits.outputWaiterCount(SESSION_ID)).toBe(0); + }, 10_000); + + it('an oversized timeout clamps rather than 400ing', async () => { + const { app, ctx } = await harness(); + // Resolve from the buffer so the assertion is about the clamp, not a real wait. + ctx.sessions.get(SESSION_ID)!.terminalBuffer = 'ALREADY_THERE\n'; + const res = await app.inject({ + method: 'GET', + url: `${URL}?match=ALREADY_THERE&from=buffer&timeout=99999999`, + }); + + expect(res.statusCode).toBe(200); + expect(res.json().data.wait.timeoutMs).toBe(MAX_WAIT_MS); + }); +}); diff --git a/test/routes/session-wait-routes.test.ts b/test/routes/session-wait-routes.test.ts new file mode 100644 index 00000000..155efb97 --- /dev/null +++ b/test/routes/session-wait-routes.test.ts @@ -0,0 +1,811 @@ +/** + * @fileoverview Route tests for `GET /api/sessions/:id/wait`. + * + * The contract this pins is the one an orchestrating agent depends on: + * - a timeout is a 200 with `wait.timedOut: true`, never a 4xx, because callers loop + * over short waits and every poll boundary would otherwise look like a failure; + * - the result is nested under `data.wait` on ALL THREE wait endpoints, so a single + * client helper reads any of them; + * - the EFFECTIVE timeout is echoed, so a caller that asked for 30 minutes and was + * clamped to 10 can tell a poll boundary from a wedged worker; + * - an unknown `until` token is a 400 rather than a silent fallback to the default, + * so a typo can never leave an agent believing it is waiting for something else, + * and a schema 400 names the parameter it rejected; + * - `stop`/`blocked` are rejected for modes that install no hooks when asked for + * EXPLICITLY, but silently dropped from the DEFAULT set, so omitting `until` never + * 400s — and `shell` counts as such a mode, even though it is not an external CLI; + * - a client that hangs up frees its waiter immediately, or a loop of + * `curl --max-time` calls wedges a process-wide cap nobody else can use; + * - a session with no PTY answers `exit`, never `idle`. + * + * Plan: docs/agent-control-plan.md + */ +import { describe, it, expect, afterEach, vi } from 'vitest'; +import fastifyCookie from '@fastify/cookie'; +import Fastify, { type FastifyInstance } from 'fastify'; +import type { ServerResponse } from 'node:http'; +import { + registerSessionRoutes, + _resetPaneLivenessState, + _paneDeathWatcherCount, +} from '../../src/web/routes/session-routes.js'; +import { createSessionListeners, attachSessionListeners } from '../../src/web/session-listener-wiring.js'; +import { installRouteErrorHandler } from '../../src/web/route-error-handler.js'; +import { ApiErrorCode, httpStatusForErrorCode } from '../../src/types.js'; +import { createMockRouteContext, type MockRouteContext } from '../mocks/index.js'; +import { sessionWaits } from '../../src/web/session-wait-registry.js'; +import { MAX_WAIT_MS, MIN_WAIT_MS, MAX_WAITERS_TOTAL, MAX_WAITERS_PER_OWNER } from '../../src/config/agent-wait.js'; + +// Distinct per file on purpose: the three wait suites share the process-wide +// `sessionWaits` singleton, so a common id let one file's leftover waiter be counted +// by another's assertion. Failed only in a 5-file run, which is how CI runs them. +const SESSION_ID = 'wait-routes-session'; + +/** Session ids a test parked filler waiters on, so cleanup can release them. */ +const fillerIds = new Set(); + +/** Park a waiter directly on the shared registry (cap tests), tracked for cleanup. */ +function fillWaiter(id: string, owner?: string): Promise { + fillerIds.add(id); + return sessionWaits.waitForSignal(id, { until: ['stop'], timeoutMs: 30_000, owner }); +} + +afterEach(() => { + // The routes use the process-wide registry; never leak a waiter into the next test. + // Deliberately NOT `cancelEverything()`: it latches the registry's stopped flag + // (one-way by design, so a request landing mid-shutdown cannot register a waiter + // nothing will ever cancel), which would leave every later test in this file + // talking to a dead registry and passing vacuously. + for (const id of [SESSION_ID, ...fillerIds]) sessionWaits.cancelAll(id); + fillerIds.clear(); + // Pane-liveness state is module-level (one cache, one watcher per pane), so it has + // to be reset or a cached probe leaks into the next test. + _resetPaneLivenessState(); + delete process.env.CODEMAN_MULTIUSER; +}); + +/** + * The shared route harness returns handler payloads verbatim, so a `{success:false}` + * body would still be HTTP 200. Production maps errorCode to status in server.ts, so + * mirror that here or every negative case passes vacuously. + * + * `rawReplies` collects each request's `reply.raw` — the RESPONSE, which is what the + * handler's hang-up detection listens on, and deliberately not `req.raw` (on a POST + * that one closes as soon as the body has been read, so wiring an abort to it kills + * every send-and-wait; see the real-HTTP suite in session-input-wait.test.ts). + * Emitting the event by hand is the only way to simulate a hang-up here at all: + * `app.inject()` never emits `close` on its own, verified. + */ +async function harness(options?: { authUser?: { username: string; role: 'admin' | 'user' } }): Promise<{ + app: FastifyInstance; + ctx: MockRouteContext; + rawReplies: ServerResponse[]; +}> { + const app = Fastify({ logger: false }); + await app.register(fastifyCookie); + const ctx = createMockRouteContext({ sessionId: SESSION_ID }); + const rawReplies: ServerResponse[] = []; + + const authUser = options?.authUser; + app.addHook('onRequest', async (req, reply) => { + // The RESPONSE, because that is what the handler's hang-up detection listens to. + rawReplies.push(reply.raw); + if (authUser) (req as unknown as { authUser: typeof authUser }).authUser = authUser; + }); + + registerSessionRoutes(app, ctx as never); + + app.addHook('preSerialization', (req, reply, payload: unknown, done) => { + const p = payload as { success?: unknown; errorCode?: unknown } | null; + if (p && typeof p === 'object' && p.success === false && reply.statusCode === 200) { + if (typeof p.errorCode === 'string') reply.code(httpStatusForErrorCode(p.errorCode as ApiErrorCode)); + } + return done(null, payload); + }); + + installRouteErrorHandler(app); + await app.ready(); + return { app, ctx, rawReplies }; +} + +describe('GET /api/sessions/:id/wait', () => { + it('resolves immediately when the session is already in a requested state', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle` }); + + expect(res.statusCode).toBe(200); + const body = res.json(); + expect(body.success).toBe(true); + expect(body.data.wait.signal).toBe('idle'); + expect(body.data.wait.immediate).toBe(true); + expect(body.data.wait.timedOut).toBe(false); + expect(body.data.wait.aborted).toBe(false); + expect(body.data.sessionId).toBe(SESSION_ID); + expect(body.data.wait.until).toEqual(['idle']); + }); + + it('nests the result under data.wait, so one client helper reads all three endpoints', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle` }); + + const { data } = res.json(); + // Session-level facts stay at the top; everything about the wait is inside it. + expect(Object.keys(data).sort()).toEqual(['limitPaused', 'sessionId', 'status', 'wait']); + expect(data.signal).toBeUndefined(); + }); + + it('reports the post-wait status and the limit-pause hint', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle` }); + + const body = res.json(); + expect(body.data.status).toBe('idle'); + expect(body.data.limitPaused).toBe(false); + }); + + it('sends Cache-Control: no-store, so a polled long-poll cannot be served from a cache', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle` }); + + expect(res.headers['cache-control']).toBe('no-store'); + }); + + it('defaults to stop,idle,exit when until is omitted', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait` }); + + const body = res.json(); + expect(body.data.wait.until).toEqual(['stop', 'idle', 'exit']); + // The mock session is idle, so the default set resolves right away. + expect(body.data.wait.signal).toBe('idle'); + }); + + it('rejects an unknown until token instead of falling back to the default', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stpo` }); + + expect(res.statusCode).toBe(400); + const body = res.json(); + expect(body.success).toBe(false); + expect(body.errorCode).toBe('INVALID_INPUT'); + expect(body.error).toContain('stpo'); + }); + + it('rejects a non-numeric timeout', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?timeout=soon` }); + + expect(res.statusCode).toBe(400); + expect(res.json().errorCode).toBe('INVALID_INPUT'); + }); + + it('names the parameter it rejected, instead of a bare "invalid parameters"', async () => { + // An agent driving this with no docs in context can only recover if the error + // says WHICH parameter was wrong; the old message named none of them. + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?timeout=30s` }); + + const body = res.json(); + expect(body.error).toContain('timeout'); + expect(body.error).toContain('wait'); + }); + + it('404s an unknown session', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: '/api/sessions/nope/wait?until=idle' }); + + expect(res.statusCode).toBe(404); + expect(res.json().success).toBe(false); + }); + + it('resolves an in-flight wait when the signal arrives', async () => { + const { app } = await harness(); + // `stop` is not the session's current signal, so this blocks. + const pending = app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop` }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(1); + sessionWaits.notifySignal(SESSION_ID, 'stop'); + + const body = (await pending).json(); + expect(body.data.wait.signal).toBe('stop'); + expect(body.data.wait.immediate).toBe(false); + expect(body.data.wait.timedOut).toBe(false); + expect(body.data.wait.waitedMs).toBeGreaterThanOrEqual(0); + }); + + it('fresh=1 waits for the next transition instead of answering from current state', async () => { + const { app } = await harness(); + const pending = app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle&fresh=1` }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(1); + sessionWaits.notifySignal(SESSION_ID, 'idle'); + + const body = (await pending).json(); + expect(body.data.wait.signal).toBe('idle'); + expect(body.data.wait.immediate).toBe(false); + }); + + it('resolves with ended when the session goes away mid-wait', async () => { + const { app } = await harness(); + const pending = app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop` }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + sessionWaits.cancelAll(SESSION_ID); + + const body = (await pending).json(); + expect(body.success).toBe(true); + expect(body.data.wait.ended).toBe(true); + expect(body.data.wait.signal).toBeNull(); + expect(body.data.wait.timedOut).toBe(false); + }); + + it('answers 200 with timedOut on timeout, never an error status', async () => { + const { app } = await harness(); + // Clamped up to the 1s floor, so this is the one deliberately slow case. + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop&timeout=1` }); + + expect(res.statusCode).toBe(200); + const body = res.json(); + expect(body.success).toBe(true); + expect(body.data.wait.timedOut).toBe(true); + expect(body.data.wait.signal).toBeNull(); + expect(body.data.wait.ended).toBe(false); + }); +}); + +describe('GET /api/sessions/:id/wait: the effective timeout is observable', () => { + it('echoes the clamped-down value when the caller asks for more than the ceiling', async () => { + // Asked for 30 minutes, got MAX_WAIT_MS. Without the echo the caller reads a + // 10-minute timeout as "30 minutes elapsed with no stop" and kills a healthy worker. + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle&timeout=1800000` }); + + expect(res.json().data.wait.timeoutMs).toBe(MAX_WAIT_MS); + }); + + it('echoes the clamped-up value at the floor too', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle&timeout=1` }); + + expect(res.json().data.wait.timeoutMs).toBe(MIN_WAIT_MS); + }); + + it('reports the applied default when timeout is omitted', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle` }); + + expect(res.json().data.wait.timeoutMs).toBeGreaterThanOrEqual(MIN_WAIT_MS); + expect(res.json().data.wait.timeoutMs).toBeLessThanOrEqual(MAX_WAIT_MS); + }); +}); + +describe('GET /api/sessions/:id/wait: repeated query parameters', () => { + it('accepts ?until=stop&until=exit, the way most clients express a list', async () => { + // Fastify delivers a repeated parameter as an array and parseWaitSignals has + // always handled one; only the schema was rejecting it. + const { app } = await harness(); + const pending = app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop&until=exit` }); + + await new Promise((resolve) => setTimeout(resolve, 20)); + sessionWaits.notifySignal(SESSION_ID, 'exit'); + + const body = (await pending).json(); + expect(body.data.wait.until).toEqual(['stop', 'exit']); + expect(body.data.wait.signal).toBe('exit'); + }); + + it('still reports an unknown token inside a repeated parameter', async () => { + const { app } = await harness(); + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop&until=stpo` }); + + expect(res.statusCode).toBe(400); + expect(res.json().error).toContain('stpo'); + }); +}); + +describe('GET /api/sessions/:id/wait: a client that hangs up frees its waiter', () => { + it('removes the waiter when the RESPONSE socket closes early', async () => { + // `curl --max-time 30 ".../wait?timeout=600000"` abandons a live waiter every + // iteration of the documented loop; sixteen of those and an innocent session + // reports busy. + // + // The response body is unreadable afterwards (the injected response really is + // destroyed, exactly as a hung-up socket would be), so the freed slot is all this + // can assert. The full behaviour, including `aborted: true` on the wire for the + // caller that did NOT hang up, is pinned over real HTTP in session-input-wait.test.ts. + const { app, rawReplies } = await harness(); + const pending = app + .inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop&timeout=600000` }) + .then( + () => 'completed', + () => 'destroyed' + ); + + await new Promise((resolve) => setTimeout(resolve, 20)); + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(1); + + rawReplies[rawReplies.length - 1].emit('close'); + await new Promise((resolve) => setTimeout(resolve, 10)); + + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(0); + expect(await pending).toBe('destroyed'); + }); + + it('a close AFTER the wait resolved changes nothing', async () => { + const { app, rawReplies } = await harness(); + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle` }); + + expect(res.json().data.wait.aborted).toBe(false); + // Node fires `close` on every completed response too, not only on a hang-up; + // `writableFinished` is what separates them. + rawReplies[rawReplies.length - 1].emit('close'); + expect(sessionWaits.totalWaiterCount()).toBe(0); + }); +}); + +describe('GET /api/sessions/:id/wait: capacity errors name the cap that was hit', () => { + it('maps the per-session cap to SESSION_BUSY / 409', async () => { + const { app } = await harness(); + const pendings = []; + for (let i = 0; i < 16; i++) { + pendings.push(app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop` })); + } + await new Promise((resolve) => setTimeout(resolve, 30)); + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(16); + + const overflow = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop` }); + expect(overflow.statusCode).toBe(409); + expect(overflow.json().errorCode).toBe('SESSION_BUSY'); + expect(overflow.json().error).toContain('session'); + + sessionWaits.cancelAll(SESSION_ID); + await Promise.all(pendings); + }); + + it('maps the process-wide cap to RATE_LIMITED / 429, because this session is not the problem', async () => { + // Reported as SESSION_BUSY, an agent concludes the session it asked about is + // busy, switches to another, and gets the identical error. + const { app } = await harness(); + const others: Promise[] = []; + for (let i = 0; i < MAX_WAITERS_TOTAL; i++) others.push(fillWaiter(`unrelated-${i}`)); + expect(sessionWaits.totalWaiterCount()).toBe(MAX_WAITERS_TOTAL); + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop` }); + expect(res.statusCode).toBe(429); + expect(res.json().errorCode).toBe('RATE_LIMITED'); + expect(res.json().error).toContain('total'); + + for (const id of fillerIds) sessionWaits.cancelAll(id); + await Promise.all(others); + }); + + it('maps the per-owner cap to RATE_LIMITED / 429 and charges the request to its user', async () => { + // Also proves the route passes ownerFor(req): without it the owner cap can never + // trip, and one user could hold the whole process-wide pool. + process.env.CODEMAN_MULTIUSER = '1'; + const { app } = await harness({ authUser: { username: 'alice', role: 'admin' } }); + + const pending = app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop` }); + await new Promise((resolve) => setTimeout(resolve, 20)); + expect(sessionWaits.ownerWaiterCount('alice')).toBe(1); + + const others: Promise[] = []; + while (sessionWaits.ownerWaiterCount('alice') < MAX_WAITERS_PER_OWNER) { + others.push(fillWaiter(`alice-${others.length}`, 'alice')); + } + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop` }); + expect(res.statusCode).toBe(429); + expect(res.json().error).toContain('owner'); + + sessionWaits.cancelAll(SESSION_ID); + for (const id of fillerIds) sessionWaits.cancelAll(id); + await Promise.all([pending, ...others]); + }); +}); + +describe('GET /api/sessions/:id/wait: modes that install no hooks', () => { + it('rejects an explicit stop, which no external CLI ever emits', async () => { + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.mode = 'codex'; + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop` }); + expect(res.statusCode).toBe(400); + expect(res.json().error).toContain('codex'); + }); + + it('rejects an explicit blocked too', async () => { + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.mode = 'opencode'; + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=blocked` }); + expect(res.statusCode).toBe(400); + }); + + it('rejects stop for a SHELL session, which is not an external CLI but installs no hooks either', async () => { + // A plain bash PTY never POSTs a Stop hook, so this was a guaranteed ten-minute + // hang dressed up as a timeout — the exact failure the guard exists to prevent. + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.mode = 'shell'; + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop` }); + expect(res.statusCode).toBe(400); + expect(res.json().error).toContain('shell'); + }); + + it('silently drops hook-only signals from the DEFAULT set instead of 400ing', async () => { + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.mode = 'gemini'; + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait` }); + expect(res.statusCode).toBe(200); + expect(res.json().data.wait.until).toEqual(['idle', 'exit']); + }); + + it('drops them from the default set for shell too', async () => { + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.mode = 'shell'; + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait` }); + expect(res.statusCode).toBe(200); + expect(res.json().data.wait.until).toEqual(['idle', 'exit']); + }); + + it('still accepts idle and exit explicitly', async () => { + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.mode = 'antigravity'; + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle,exit` }); + expect(res.statusCode).toBe(200); + expect(res.json().data.wait.until).toEqual(['idle', 'exit']); + }); +}); + +describe('GET /api/sessions/:id/wait: liveness beats the reported status', () => { + it('answers exit for a session whose PTY is gone, not the idle its status claims', async () => { + // Session parks a dead PTY at status 'idle' and the object survives in the map, + // so the DEFAULT wait used to answer {signal:"idle", immediate:true} for a + // crashed worker — 200, success, no error, and the agent prompts a corpse. + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + session.pid = null; + session.status = 'idle'; + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait` }); + const body = res.json(); + expect(body.data.wait.signal).toBe('exit'); + expect(body.data.wait.immediate).toBe(true); + // The raw status is still reported, so nothing is hidden from the caller. + expect(body.data.status).toBe('idle'); + }); + + it('resolves until=exit immediately for an already-exited session instead of blocking', async () => { + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.pid = null; + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=exit&timeout=1` }); + expect(res.json().data.wait.signal).toBe('exit'); + expect(res.json().data.wait.timedOut).toBe(false); + }); + + it('does not report idle for a dead session even when idle was asked for explicitly', async () => { + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.pid = null; + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle&timeout=1` }); + expect(res.json().data.wait.signal).toBeNull(); + expect(res.json().data.wait.timedOut).toBe(true); + }); + + it('a live busy session resolves until=working immediately', async () => { + // Unreachable before: MockSession used 'working', which is not a SessionStatus, + // so signalForStatus fell through to null and this branch had no coverage. + const { app, ctx } = await harness(); + ctx.sessions.get(SESSION_ID)!.status = 'busy'; + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=working` }); + expect(res.json().data.wait.signal).toBe('working'); + expect(res.json().data.wait.immediate).toBe(true); + }); + + it('a live stopped/error session maps to exit', async () => { + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + + session.status = 'stopped'; + const stopped = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=exit` }); + expect(stopped.json().data.wait.signal).toBe('exit'); + + session.status = 'error'; + const errored = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=exit` }); + expect(errored.json().data.wait.signal).toBe('exit'); + }); +}); + +/** + * The other half of the wait contract, exercised through the real listener wiring + * rather than a route: a PTY that dies must RELEASE every waiter, not only the ones + * that asked for `exit`. + * + * It lives in this file because it pins the same promise the routes above make + * ("never hang"), and because the failure is only visible from the caller's side: + * the exit handler detaches the `terminal`, `idle` and `working` listeners moments + * later, so anything still registered afterwards is waiting on feeds that no longer + * exist and can only time out. + */ +describe('a PTY exit releases waiters that did not ask for exit', () => { + /** Everything the exit handler touches; the wait release must not depend on any of it. */ + function stubDeps(overrides: Record = {}) { + return { + broadcast: vi.fn(), + batchTerminalData: vi.fn(), + batchTaskUpdate: vi.fn(), + broadcastSessionStateDebounced: vi.fn(), + sendPushNotifications: vi.fn(), + persistSessionState: vi.fn(), + getSessionStateWithRespawn: vi.fn(() => ({})), + getRunSummaryTracker: vi.fn(() => undefined), + stopTranscriptWatcher: vi.fn(), + cleanupSessionBatches: vi.fn(), + cancelPersistDebounce: vi.fn(), + removeRunSummaryTracker: vi.fn(), + removeSessionListenerRefs: vi.fn(), + cleanupRespawnOnExit: vi.fn(), + getStore: vi.fn(() => ({ updateRalphState: vi.fn() })), + registerAttachment: vi.fn(async () => {}), + ...overrides, + }; + } + + it('answers an until=working waiter with ended instead of leaving it to time out', async () => { + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + attachSessionListeners(session as never, createSessionListeners(session as never, stubDeps() as never)); + + const pending = app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=working&timeout=600000` }); + await new Promise((resolve) => setTimeout(resolve, 20)); + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(1); + + session.emit('exit', 1); + + const body = (await pending).json(); + expect(body.data.wait.ended).toBe(true); + expect(body.data.wait.timedOut).toBe(false); + expect(body.data.wait.signal).toBeNull(); + }); + + it('still gives an until=exit waiter its signal, not a bare ended', async () => { + // Ordering matters: notifySignal('exit') must run BEFORE cancelAll, or a caller + // that asked the right question gets the generic answer. + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + attachSessionListeners(session as never, createSessionListeners(session as never, stubDeps() as never)); + + const pending = app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=exit&fresh=1` }); + await new Promise((resolve) => setTimeout(resolve, 20)); + + session.emit('exit', 0); + + const body = (await pending).json(); + expect(body.data.wait.signal).toBe('exit'); + expect(body.data.wait.ended).toBe(false); + }); + + it('releases output waiters too, whose only feed the exit handler is about to detach', async () => { + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + attachSessionListeners(session as never, createSessionListeners(session as never, stubDeps() as never)); + + const pending = app.inject({ + method: 'GET', + url: `/api/sessions/${SESSION_ID}/wait-output?match=DONE&timeout=600000`, + }); + await new Promise((resolve) => setTimeout(resolve, 20)); + expect(sessionWaits.outputWaiterCount(SESSION_ID)).toBe(1); + + session.emit('exit', 1); + + const body = (await pending).json(); + expect(body.data.wait.ended).toBe(true); + expect(body.data.wait.matched).toBe(false); + expect(sessionWaits.waiterCount(SESSION_ID)).toBe(0); + }); + + it('releases them even when a later step of the exit handler throws', async () => { + // Which is why the release is the first thing in the handler. + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + const deps = stubDeps({ + broadcast: vi.fn(() => { + throw new Error('SSE is down'); + }), + }); + attachSessionListeners(session as never, createSessionListeners(session as never, deps as never)); + + const pending = app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop&timeout=600000` }); + await new Promise((resolve) => setTimeout(resolve, 20)); + + session.emit('exit', 1); + + expect((await pending).json().data.wait.ended).toBe(true); + }); +}); + +/** + * Worker liveness for a tmux-backed session. + * + * `session.pid` is the local `tmux attach` CLIENT, not the worker. Codeman sets + * `remain-on-exit on`, so when the command inside the pane exits tmux keeps the pane + * (`pane_dead=1`), the tmux session survives, the attach client keeps running, `pid` + * never goes null and NO exit event fires. Reproduced live on a shell worker killed + * with `exit 42`: tmux said `pane_dead=1 status=42` while Codeman said + * `pid=309406 status=idle` and the default wait answered + * `{signal:"idle", immediate:true, waitedMs:0}` for a corpse. + * + * These cases could not exist before, because `MockSession.pid` is set by hand: the + * `pid === null` branch is the one production never reaches. + */ +describe('GET /api/sessions/:id/wait: a dead tmux worker', () => { + /** Mock ctx doubles carry no `isPaneDead`; the route treats that as "cannot tell". */ + function setPaneDead(ctx: MockRouteContext, dead: boolean): ReturnType { + const probe = vi.fn(() => dead); + (ctx.mux as unknown as { isPaneDead: (name: string) => boolean }).isPaneDead = probe as never; + return probe; + } + + it('answers exit, not the idle the session still reports', async () => { + const { app, ctx } = await harness(); + const session = ctx.sessions.get(SESSION_ID)!; + setPaneDead(ctx, true); + // Exactly the live state: attach client alive, status idle, worker gone. + expect(session.pid).not.toBeNull(); + expect(session.status).toBe('idle'); + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait` }); + const body = res.json(); + expect(body.data.wait.signal).toBe('exit'); + expect(body.data.wait.immediate).toBe(true); + // The raw status is still reported, so nothing is hidden from the caller. + expect(body.data.status).toBe('idle'); + }); + + it('resolves until=exit immediately instead of burning the whole timeout', async () => { + const { app, ctx } = await harness(); + setPaneDead(ctx, true); + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=exit&timeout=1` }); + expect(res.json().data.wait.signal).toBe('exit'); + expect(res.json().data.wait.timedOut).toBe(false); + }); + + it('does not answer idle for a dead worker even when idle was asked for explicitly', async () => { + const { app, ctx } = await harness(); + setPaneDead(ctx, true); + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle&timeout=1` }); + expect(res.json().data.wait.signal).toBeNull(); + expect(res.json().data.wait.timedOut).toBe(true); + }); + + it('a live pane is unaffected', async () => { + const { app, ctx } = await harness(); + setPaneDead(ctx, false); + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle` }); + expect(res.json().data.wait.signal).toBe('idle'); + }); + + it('caches the probe, so a poll loop cannot exec tmux once per request', async () => { + const { app, ctx } = await harness(); + const probe = setPaneDead(ctx, false); + + for (let i = 0; i < 10; i++) { + await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle` }); + } + expect(probe.mock.calls.length).toBeLessThanOrEqual(2); + }); + + it('never probes a session that is not tmux-backed', async () => { + const { app, ctx } = await harness(); + const probe = setPaneDead(ctx, true); + ctx.sessions.get(SESSION_ID)!.usesMux = false; + + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=idle` }); + expect(probe).not.toHaveBeenCalled(); + // Falls back to the pid rule, which is the right one for a direct PTY. + expect(res.json().data.wait.signal).toBe('idle'); + }); + + it('releases a wait when the worker dies WHILE it is parked', async () => { + // The common orchestration case, and the one a request-time probe cannot see: no + // exit event, no output, nothing — the caller would block for its full timeout. + const { app, ctx } = await harness(); + let dead = false; + (ctx.mux as unknown as { isPaneDead: () => boolean }).isPaneDead = () => dead; + + const pending = app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop&timeout=600000` }); + await new Promise((resolve) => setTimeout(resolve, 50)); + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(1); + expect(_paneDeathWatcherCount()).toBe(1); + + dead = true; + const body = (await pending).json(); + expect(body.data.wait.ended).toBe(true); + expect(body.data.wait.timedOut).toBe(false); + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(0); + // ...and the watcher is torn down with the last waiter that needed it. + expect(_paneDeathWatcherCount()).toBe(0); + }, 10_000); + + it('an until=exit caller parked when the worker dies gets its signal, not a bare ended', async () => { + const { app, ctx } = await harness(); + let dead = false; + (ctx.mux as unknown as { isPaneDead: () => boolean }).isPaneDead = () => dead; + + const pending = app.inject({ + method: 'GET', + url: `/api/sessions/${SESSION_ID}/wait?until=exit&fresh=1&timeout=600000`, + }); + await new Promise((resolve) => setTimeout(resolve, 50)); + dead = true; + + const body = (await pending).json(); + expect(body.data.wait.signal).toBe('exit'); + expect(body.data.wait.ended).toBe(false); + }, 10_000); + + it('starts no watcher at all when the session is not tmux-backed', async () => { + const { app, ctx } = await harness(); + setPaneDead(ctx, false); + ctx.sessions.get(SESSION_ID)!.usesMux = false; + + const pending = app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop&timeout=600000` }); + await new Promise((resolve) => setTimeout(resolve, 30)); + expect(_paneDeathWatcherCount()).toBe(0); + + sessionWaits.cancelAll(SESSION_ID); + await pending; + }); + + it('shares ONE watcher across every wait parked on the same session', async () => { + const { app, ctx } = await harness(); + setPaneDead(ctx, false); + + const pendings = []; + for (let i = 0; i < 5; i++) { + pendings.push(app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?until=stop&timeout=600000` })); + } + await new Promise((resolve) => setTimeout(resolve, 40)); + expect(sessionWaits.signalWaiterCount(SESSION_ID)).toBe(5); + expect(_paneDeathWatcherCount()).toBe(1); + + sessionWaits.cancelAll(SESSION_ID); + await Promise.all(pendings); + expect(_paneDeathWatcherCount()).toBe(0); + }); +}); + +describe('GET /api/sessions/:id/wait: an oversized timeout clamps, it does not 400', () => { + it('accepts a value above the old schema ceiling and reports the clamp', async () => { + // "Clamped to [1000, 600000]" has to mean it: `timeout=600001` clamping while + // `timeout=99999999` 400s is the same documented rule producing two outcomes. + const { app } = await harness(); + const res = await app.inject({ + method: 'GET', + url: `/api/sessions/${SESSION_ID}/wait?until=idle&timeout=99999999`, + }); + + expect(res.statusCode).toBe(200); + expect(res.json().data.wait.timeoutMs).toBe(MAX_WAIT_MS); + }); + + it('still rejects a non-finite or non-integer timeout', async () => { + const { app } = await harness(); + for (const value of ['1e999', 'soon', '-1', '1.5']) { + const res = await app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/wait?timeout=${value}` }); + expect(res.statusCode, `timeout=${value}`).toBe(400); + } + }); +}); diff --git a/test/session-cli-builder.test.ts b/test/session-cli-builder.test.ts index f853d0ee..3cb0db3b 100644 --- a/test/session-cli-builder.test.ts +++ b/test/session-cli-builder.test.ts @@ -64,3 +64,38 @@ describe('buildMuxAttachEnv', () => { } }); }); + +describe('spawn env CODEMAN_API_URL (no fallback)', () => { + const withApiUrl = (value: string | undefined, fn: () => void) => { + const original = process.env.CODEMAN_API_URL; + if (value === undefined) delete process.env.CODEMAN_API_URL; + else process.env.CODEMAN_API_URL = value; + try { + fn(); + } finally { + if (original === undefined) delete process.env.CODEMAN_API_URL; + else process.env.CODEMAN_API_URL = original; + } + }; + + it('passes the server-stamped URL through verbatim', async () => { + const { buildClaudeEnv, buildShellEnv } = await import('../src/session-cli-builder.js'); + withApiUrl('https://127.0.0.1:3199', () => { + expect(buildClaudeEnv('test-session').CODEMAN_API_URL).toBe('https://127.0.0.1:3199'); + expect(buildShellEnv('test-session').CODEMAN_API_URL).toBe('https://127.0.0.1:3199'); + }); + }); + + // A hardcoded fallback was the wrong scheme on HTTPS installs. The key must be + // genuinely ABSENT when unset: present-with-undefined would serialize through + // node-pty as the literal string "CODEMAN_API_URL=undefined" (COD-115). + it('leaves the key absent (not undefined, not a fallback) when the server has not stamped one', async () => { + const { buildClaudeEnv, buildShellEnv } = await import('../src/session-cli-builder.js'); + withApiUrl(undefined, () => { + for (const env of [buildClaudeEnv('test-session'), buildShellEnv('test-session')]) { + expect('CODEMAN_API_URL' in env).toBe(false); + expect(JSON.stringify(env)).not.toContain('localhost:3000'); + } + }); + }); +}); diff --git a/test/session-wait-registry.test.ts b/test/session-wait-registry.test.ts new file mode 100644 index 00000000..46da5141 --- /dev/null +++ b/test/session-wait-registry.test.ts @@ -0,0 +1,1055 @@ +/** + * @fileoverview Unit tests for the blocking-wait registry that backs the agent + * wait primitives (`GET /api/sessions/:id/wait`, `.../wait-output`, and the + * `wait` field on `POST .../input`). Plan: `docs/agent-control-plan.md`. + * + * The registry holds no IO and no Session reference, so everything here runs + * against real (short) timers with no mocks beyond one clearTimeout spy. + */ +import { describe, it, expect, vi } from 'vitest'; +import { + SessionWaitRegistry, + WaitCapacityError, + hooksAvailableForMode, + parseWaitSignals, + resolveWaitSignals, + signalForStatus, + DEFAULT_WAIT_SIGNALS, + WAIT_SIGNALS, +} from '../src/web/session-wait-registry.js'; +import type { SessionMode } from '../src/types.js'; + +const sleep = (ms: number) => new Promise((resolve) => setTimeout(resolve, ms)); + +describe('parseWaitSignals', () => { + it('parses a comma-separated string', () => { + expect(parseWaitSignals('stop,idle')).toEqual({ signals: ['stop', 'idle'], invalid: [] }); + }); + + it('parses an array, including comma-joined entries', () => { + expect(parseWaitSignals(['stop', 'idle,exit'])).toEqual({ + signals: ['stop', 'idle', 'exit'], + invalid: [], + }); + }); + + it('trims and lowercases', () => { + expect(parseWaitSignals(' STOP , Idle ')).toEqual({ signals: ['stop', 'idle'], invalid: [] }); + }); + + it('dedups, first occurrence wins', () => { + expect(parseWaitSignals('idle,stop,idle')).toEqual({ signals: ['idle', 'stop'], invalid: [] }); + }); + + it('reports invalid tokens instead of silently dropping them', () => { + // The whole point: a typo must surface as a 400, never as "waiting for the default". + const parsed = parseWaitSignals('stop,stpo,done'); + expect(parsed.signals).toEqual(['stop']); + expect(parsed.invalid).toEqual(['stpo', 'done']); + }); + + it('dedups invalid tokens too', () => { + expect(parseWaitSignals('nope,nope').invalid).toEqual(['nope']); + }); + + it('returns empty for absent / non-string input', () => { + expect(parseWaitSignals(undefined)).toEqual({ signals: [], invalid: [] }); + expect(parseWaitSignals(null)).toEqual({ signals: [], invalid: [] }); + expect(parseWaitSignals(42)).toEqual({ signals: [], invalid: [] }); + expect(parseWaitSignals('')).toEqual({ signals: [], invalid: [] }); + expect(parseWaitSignals(',, ,')).toEqual({ signals: [], invalid: [] }); + }); + + it('accepts every documented signal', () => { + expect(parseWaitSignals(WAIT_SIGNALS.join(','))).toEqual({ + signals: [...WAIT_SIGNALS], + invalid: [], + }); + }); + + it('the default set is itself valid', () => { + expect(parseWaitSignals(DEFAULT_WAIT_SIGNALS.join(','))).toEqual({ + signals: [...DEFAULT_WAIT_SIGNALS], + invalid: [], + }); + }); + + it('the default set includes exit, so a crashed worker never burns the full timeout', () => { + expect(DEFAULT_WAIT_SIGNALS).toContain('exit'); + }); +}); + +describe('signalForStatus', () => { + it('maps live statuses', () => { + expect(signalForStatus('idle')).toBe('idle'); + expect(signalForStatus('busy')).toBe('working'); + }); + + it('maps both dead statuses to exit, so a caller that sees one does not hang', () => { + // Narrower than it looks, and the JSDoc says so: a PTY that merely exits parks the + // session at 'idle', so 'stopped' is reached only on a spawn failure or an explicit + // stop(killMux:true). This mapping is right; it is not a liveness check. + expect(signalForStatus('stopped')).toBe('exit'); + expect(signalForStatus('error')).toBe('exit'); + }); + + it('cannot distinguish a finished session from a dead one', () => { + // Pinning the documented limitation: both PTY onExit handlers set 'idle', so this + // is what a caller sees for a crashed worker. Liveness must come from elsewhere. + expect(signalForStatus('idle')).toBe('idle'); + }); +}); + +describe('hooksAvailableForMode', () => { + it('is true only for claude', () => { + expect(hooksAvailableForMode('claude')).toBe(true); + }); + + it('is false for shell, which installs no hooks despite not being an external CLI', () => { + // The bug this replaced keyed off isExternalCliMode(), which excludes shell, so + // until=stop on a bash PTY was accepted and then blocked for the full timeout. + expect(hooksAvailableForMode('shell')).toBe(false); + }); + + it('is false for every external CLI mode', () => { + for (const mode of ['opencode', 'codex', 'gemini', 'antigravity'] as const) { + expect(hooksAvailableForMode(mode)).toBe(false); + } + }); +}); + +describe('resolveWaitSignals', () => { + const claude = { mode: 'claude' as SessionMode }; + const codex = { mode: 'codex' as SessionMode }; + const shell = { mode: 'shell' as SessionMode }; + + it('returns the explicit set for a normal request', () => { + expect(resolveWaitSignals('stop,exit', claude)).toEqual({ until: ['stop', 'exit'], error: null }); + }); + + it('falls back to the default set when nothing is asked for', () => { + expect(resolveWaitSignals(undefined, claude)).toEqual({ until: ['stop', 'idle', 'exit'], error: null }); + // `wait: true` on the input route arrives here as undefined, same path. + expect(resolveWaitSignals('', claude).until).toEqual(['stop', 'idle', 'exit']); + }); + + it('errors on an unknown token rather than silently defaulting', () => { + const result = resolveWaitSignals('stop,stpo', claude); + expect(result.until).toEqual([]); + expect(result.error).toContain('stpo'); + expect(result.error).toContain('idle, working, stop, blocked, exit'); + }); + + it('errors when a hook-only signal is asked for EXPLICITLY on an external CLI', () => { + const result = resolveWaitSignals('stop', codex); + expect(result.until).toEqual([]); + expect(result.error).toContain('codex'); + expect(resolveWaitSignals('blocked', codex).error).toBeTruthy(); + expect(resolveWaitSignals('idle,stop', codex).error).toBeTruthy(); + }); + + it('drops hook-only signals from the DEFAULT set instead of erroring', () => { + // Omitting the parameter must never 400, whatever the mode. + expect(resolveWaitSignals(undefined, codex)).toEqual({ until: ['idle', 'exit'], error: null }); + }); + + it('still allows the supported signals explicitly on an external CLI', () => { + expect(resolveWaitSignals('idle,exit', codex)).toEqual({ until: ['idle', 'exit'], error: null }); + }); + + it('reports an unknown token even when the mode would also reject', () => { + // Validity is checked before mode support, so the message names the typo. + expect(resolveWaitSignals('stpo,stop', codex).error).toContain('stpo'); + }); + + it('rejects hook-only signals on a SHELL session too', () => { + // A shell session is a plain bash PTY with no Claude Code hooks, so `stop` can + // never arrive; accepting it was a guaranteed ten-minute hold by construction. + const result = resolveWaitSignals('stop', shell); + expect(result.until).toEqual([]); + expect(result.error).toContain('shell'); + expect(resolveWaitSignals('blocked', shell).error).toBeTruthy(); + }); + + it('drops hook-only signals from the DEFAULT set for shell, without erroring', () => { + expect(resolveWaitSignals(undefined, shell)).toEqual({ until: ['idle', 'exit'], error: null }); + }); + + it('names the mode in the rejection so the caller can see why', () => { + expect(resolveWaitSignals('stop', codex).error).toContain('codex'); + expect(resolveWaitSignals('stop', { mode: 'gemini' as SessionMode }).error).toContain('gemini'); + }); +}); + +describe('SessionWaitRegistry: signal waits', () => { + it('resolves immediately when the session is already in a requested state', async () => { + const reg = new SessionWaitRegistry(); + const result = await reg.waitForSignal('s1', { + until: ['idle'], + timeoutMs: 5000, + currentSignal: 'idle', + }); + expect(result).toEqual({ + signal: 'idle', + timedOut: false, + immediate: true, + ended: false, + aborted: false, + waitedMs: 0, + timeoutMs: 5000, + }); + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('does not resolve immediately when the current signal was not requested', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForSignal('s1', { + until: ['stop'], + timeoutMs: 5000, + currentSignal: 'idle', + }); + expect(reg.signalWaiterCount('s1')).toBe(1); + reg.notifySignal('s1', 'stop'); + expect((await promise).signal).toBe('stop'); + }); + + it('requireTransition ignores the current state and waits for the next one', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForSignal('s1', { + until: ['idle'], + timeoutMs: 5000, + currentSignal: 'idle', + requireTransition: true, + }); + expect(reg.signalWaiterCount('s1')).toBe(1); + + const result = await Promise.resolve().then(() => { + reg.notifySignal('s1', 'idle'); + return promise; + }); + expect(result.signal).toBe('idle'); + expect(result.immediate).toBe(false); + }); + + it('resolves on the first of several requested signals', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForSignal('s1', { until: ['stop', 'blocked'], timeoutMs: 5000 }); + reg.notifySignal('s1', 'blocked'); + const result = await promise; + expect(result.signal).toBe('blocked'); + expect(result.timedOut).toBe(false); + expect(result.ended).toBe(false); + expect(result.waitedMs).toBeGreaterThanOrEqual(0); + }); + + it('ignores signals that were not requested', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + expect(reg.notifySignal('s1', 'working')).toBe(0); + expect(reg.notifySignal('s1', 'idle')).toBe(0); + expect(reg.signalWaiterCount('s1')).toBe(1); + reg.notifySignal('s1', 'stop'); + expect((await promise).signal).toBe('stop'); + }); + + it('ignores signals for other sessions', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + expect(reg.notifySignal('s2', 'stop')).toBe(0); + expect(reg.signalWaiterCount('s1')).toBe(1); + reg.notifySignal('s1', 'stop'); + await promise; + }); + + it('wakes every waiter that asked for the signal, and only those', async () => { + const reg = new SessionWaitRegistry(); + const a = reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + const b = reg.waitForSignal('s1', { until: ['stop', 'idle'], timeoutMs: 5000 }); + const c = reg.waitForSignal('s1', { until: ['exit'], timeoutMs: 5000 }); + expect(reg.signalWaiterCount('s1')).toBe(3); + + expect(reg.notifySignal('s1', 'stop')).toBe(2); + expect((await a).signal).toBe('stop'); + expect((await b).signal).toBe('stop'); + expect(reg.signalWaiterCount('s1')).toBe(1); + + reg.notifySignal('s1', 'exit'); + expect((await c).signal).toBe('exit'); + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('times out without erroring, and reports it as a normal outcome', async () => { + const reg = new SessionWaitRegistry(); + const result = await reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 20 }); + expect(result.timedOut).toBe(true); + expect(result.signal).toBeNull(); + expect(result.ended).toBe(false); + expect(result.waitedMs).toBeGreaterThanOrEqual(0); + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('an empty until set can only time out', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForSignal('s1', { until: [], timeoutMs: 20 }); + expect(reg.notifySignal('s1', 'stop')).toBe(0); + expect((await promise).timedOut).toBe(true); + }); + + it('cancelAll resolves pending waiters with ended, never leaving them hanging', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + expect(reg.cancelAll('s1')).toBe(1); + const result = await promise; + expect(result.ended).toBe(true); + expect(result.timedOut).toBe(false); + expect(result.signal).toBeNull(); + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('honors the exit-before-cancel ordering contract', async () => { + // The wiring must notify 'exit' BEFORE cancelAll, so an until=exit caller sees + // the signal rather than a bare ended. + const reg = new SessionWaitRegistry(); + const promise = reg.waitForSignal('s1', { until: ['exit'], timeoutMs: 5000 }); + reg.notifySignal('s1', 'exit'); + expect(reg.cancelAll('s1')).toBe(0); + const result = await promise; + expect(result.signal).toBe('exit'); + expect(result.ended).toBe(false); + }); + + it('clears the timer when a signal resolves the wait', async () => { + const reg = new SessionWaitRegistry(); + const spy = vi.spyOn(globalThis, 'clearTimeout'); + const before = spy.mock.calls.length; + const promise = reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + reg.notifySignal('s1', 'stop'); + await promise; + expect(spy.mock.calls.length).toBeGreaterThan(before); + spy.mockRestore(); + }); + + it('echoes the effective timeout on every outcome', async () => { + // A caller that asked for 30 minutes and got 600s must be able to SEE that, or it + // reads the timeout as a stalled worker and kills a session that was fine. + const reg = new SessionWaitRegistry(); + expect((await reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 20 })).timeoutMs).toBe(20); + + const signalled = reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 4321 }); + reg.notifySignal('s1', 'stop'); + expect((await signalled).timeoutMs).toBe(4321); + + const cancelled = reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 777 }); + reg.cancelAll('s1'); + expect((await cancelled).timeoutMs).toBe(777); + }); +}); + +describe('SessionWaitRegistry: cancellation', () => { + it('frees the slot as soon as the caller hangs up', async () => { + // The whole point: the response can no longer be sent, so holding the waiter for + // the rest of its timeout only denies the pool to somebody else. + const reg = new SessionWaitRegistry(); + const controller = new AbortController(); + const promise = reg.waitForSignal('s1', { + until: ['stop'], + timeoutMs: 60_000, + abortSignal: controller.signal, + }); + expect(reg.signalWaiterCount('s1')).toBe(1); + + controller.abort(); + const result = await promise; + expect(result.aborted).toBe(true); + expect(result.ended).toBe(true); + expect(result.timedOut).toBe(false); + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('frees an output waiter the same way', async () => { + const reg = new SessionWaitRegistry(); + const controller = new AbortController(); + const promise = reg.waitForOutput('s1', { + match: 'never', + timeoutMs: 60_000, + abortSignal: controller.signal, + }); + expect(reg.outputWaiterCount('s1')).toBe(1); + + controller.abort(); + const result = await promise; + expect(result.aborted).toBe(true); + expect(result.ended).toBe(true); + expect(result.matched).toBe(false); + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('registers nothing at all when the signal is already aborted', () => { + const reg = new SessionWaitRegistry(); + const controller = new AbortController(); + controller.abort(); + void reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 60_000, abortSignal: controller.signal }); + void reg.waitForOutput('s1', { match: 'x', timeoutMs: 60_000, abortSignal: controller.signal }); + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('an already-aborted caller is never rejected for capacity', async () => { + // It takes no slot, so refusing it would be a lie AND would cost the caller a + // retry it cannot act on. + const reg = new SessionWaitRegistry({ maxWaitersPerSession: 1, maxWaitersTotal: 1 }); + void reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 60_000 }); + const controller = new AbortController(); + controller.abort(); + const result = await reg.waitForSignal('s1', { + until: ['stop'], + timeoutMs: 60_000, + abortSignal: controller.signal, + }); + expect(result.aborted).toBe(true); + reg.cancelEverything(); + }); + + it('aborting an already-resolved waiter is a no-op', async () => { + const reg = new SessionWaitRegistry(); + const controller = new AbortController(); + const promise = reg.waitForSignal('s1', { + until: ['stop'], + timeoutMs: 5000, + abortSignal: controller.signal, + }); + reg.notifySignal('s1', 'stop'); + const result = await promise; + expect(result.signal).toBe('stop'); + + controller.abort(); + await sleep(5); + // The settled result is unchanged and no bookkeeping went negative. + expect(result.aborted).toBe(false); + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('detaches its abort listener when the wait resolves another way', async () => { + // A caller reusing one controller across a loop of waits must not accumulate + // listeners on it for the life of the request. + const reg = new SessionWaitRegistry(); + const controller = new AbortController(); + for (let i = 0; i < 5; i++) { + const promise = reg.waitForSignal('s1', { + until: ['stop'], + timeoutMs: 5000, + abortSignal: controller.signal, + }); + reg.notifySignal('s1', 'stop'); + await promise; + } + // Node exposes the count only through the internal getter, so assert the effect: + // nothing is registered and a late abort still changes nothing. + controller.abort(); + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('aborted is false on every non-abort outcome', async () => { + const reg = new SessionWaitRegistry(); + expect((await reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 20 })).aborted).toBe(false); + expect((await reg.waitForOutput('s1', { match: 'x', timeoutMs: 20 })).aborted).toBe(false); + const immediate = await reg.waitForSignal('s1', { until: ['idle'], timeoutMs: 20, currentSignal: 'idle' }); + expect(immediate.aborted).toBe(false); + }); +}); + +describe('SessionWaitRegistry: shutdown latch', () => { + it('refuses to register a new waiter once stopped, resolving ended instead', async () => { + // server.stop() cancels waiters well before app.close(), with several awaits in + // between while the listener still accepts requests. A wait landing in that window + // used to register a timer nothing would ever cancel and stall shutdown for the + // full MAX_WAIT_MS. + const reg = new SessionWaitRegistry(); + reg.stop(); + expect(reg.isStopped).toBe(true); + + const signal = await reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 600_000 }); + expect(signal.ended).toBe(true); + expect(signal.timedOut).toBe(false); + expect(signal.aborted).toBe(false); + + const output = await reg.waitForOutput('s1', { match: 'never', timeoutMs: 600_000 }); + expect(output.ended).toBe(true); + expect(output.matched).toBe(false); + + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('stop() resolves the waiters that were already pending', async () => { + const reg = new SessionWaitRegistry(); + const pending = [ + reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 600_000 }), + reg.waitForOutput('s2', { match: 'never', timeoutMs: 600_000 }), + ]; + expect(reg.stop()).toBe(2); + for (const result of await Promise.all(pending)) expect(result.ended).toBe(true); + }); + + it('cancelEverything latches too, so the existing shutdown call site is covered', async () => { + const reg = new SessionWaitRegistry(); + reg.cancelEverything(); + expect(reg.isStopped).toBe(true); + expect((await reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 600_000 })).ended).toBe(true); + }); + + it('a fresh registry is not stopped', () => { + expect(new SessionWaitRegistry().isStopped).toBe(false); + }); +}); + +describe('SessionWaitRegistry: capacity', () => { + it('rejects past the per-session cap with scope "session"', () => { + const reg = new SessionWaitRegistry({ maxWaitersPerSession: 2, maxWaitersTotal: 100 }); + void reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + void reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + + try { + reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + expect.unreachable('third wait should have been rejected'); + } catch (err) { + expect(err).toBeInstanceOf(WaitCapacityError); + expect((err as WaitCapacityError).scope).toBe('session'); + } + + // A different session is unaffected by another session's cap. + expect(() => reg.waitForSignal('s2', { until: ['stop'], timeoutMs: 5000 })).not.toThrow(); + reg.cancelEverything(); + }); + + it('rejects past the global cap with scope "total"', () => { + const reg = new SessionWaitRegistry({ maxWaitersPerSession: 10, maxWaitersTotal: 2 }); + void reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + void reg.waitForSignal('s2', { until: ['stop'], timeoutMs: 5000 }); + + try { + reg.waitForSignal('s3', { until: ['stop'], timeoutMs: 5000 }); + expect.unreachable('third wait should have been rejected'); + } catch (err) { + expect(err).toBeInstanceOf(WaitCapacityError); + expect((err as WaitCapacityError).scope).toBe('total'); + } + reg.cancelEverything(); + }); + + it('counts both waiter kinds against the caps', () => { + const reg = new SessionWaitRegistry({ maxWaitersPerSession: 2, maxWaitersTotal: 100 }); + void reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + void reg.waitForOutput('s1', { match: 'done', timeoutMs: 5000 }); + expect(reg.waiterCount('s1')).toBe(2); + expect(() => reg.waitForOutput('s1', { match: 'x', timeoutMs: 5000 })).toThrow(WaitCapacityError); + reg.cancelEverything(); + }); + + it('frees capacity as waiters resolve', async () => { + const reg = new SessionWaitRegistry({ maxWaitersPerSession: 1, maxWaitersTotal: 100 }); + const first = reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + expect(() => reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 })).toThrow(WaitCapacityError); + reg.notifySignal('s1', 'stop'); + await first; + expect(() => reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 })).not.toThrow(); + reg.cancelEverything(); + }); + + it('an immediate resolve is never rejected by a full pool', async () => { + // An already-satisfied wait holds no resource, so it must not be capacity-checked. + const reg = new SessionWaitRegistry({ maxWaitersPerSession: 1, maxWaitersTotal: 1 }); + void reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + const result = await reg.waitForSignal('s1', { + until: ['idle'], + timeoutMs: 5000, + currentSignal: 'idle', + }); + expect(result.immediate).toBe(true); + reg.cancelEverything(); + }); + + it('rejects past the per-owner cap with scope "owner", across sessions', () => { + // One user must not be able to occupy the whole process-wide pool and deny the + // primitive to everyone else, admin included. + const reg = new SessionWaitRegistry({ maxWaitersPerSession: 10, maxWaitersPerOwner: 2, maxWaitersTotal: 100 }); + void reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000, owner: 'alice' }); + void reg.waitForOutput('s2', { match: 'x', timeoutMs: 5000, owner: 'alice' }); + expect(reg.ownerWaiterCount('alice')).toBe(2); + + try { + reg.waitForSignal('s3', { until: ['stop'], timeoutMs: 5000, owner: 'alice' }); + expect.unreachable('third wait should have been rejected'); + } catch (err) { + expect(err).toBeInstanceOf(WaitCapacityError); + expect((err as WaitCapacityError).scope).toBe('owner'); + expect((err as Error).message).toContain('owner'); + } + + // Another user is unaffected, which is the entire point. + expect(() => reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000, owner: 'bob' })).not.toThrow(); + reg.cancelEverything(); + }); + + it('ignores the owner cap when no owner is given (single-user mode)', () => { + const reg = new SessionWaitRegistry({ maxWaitersPerSession: 10, maxWaitersPerOwner: 1, maxWaitersTotal: 100 }); + for (let i = 0; i < 5; i++) { + expect(() => reg.waitForSignal(`s${i}`, { until: ['stop'], timeoutMs: 5000 })).not.toThrow(); + } + reg.cancelEverything(); + }); + + it('frees owner capacity as waiters resolve, time out and abort', async () => { + const reg = new SessionWaitRegistry({ maxWaitersPerSession: 10, maxWaitersPerOwner: 3, maxWaitersTotal: 100 }); + const resolved = reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000, owner: 'alice' }); + const timedOut = reg.waitForSignal('s2', { until: ['stop'], timeoutMs: 20, owner: 'alice' }); + const controller = new AbortController(); + const aborted = reg.waitForOutput('s3', { + match: 'x', + timeoutMs: 5000, + owner: 'alice', + abortSignal: controller.signal, + }); + expect(reg.ownerWaiterCount('alice')).toBe(3); + + reg.notifySignal('s1', 'stop'); + controller.abort(); + await Promise.all([resolved, timedOut, aborted]); + expect(reg.ownerWaiterCount('alice')).toBe(0); + expect(() => reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000, owner: 'alice' })).not.toThrow(); + reg.cancelEverything(); + }); + + it('does not double-release an owner slot when an abort races a resolve', async () => { + const reg = new SessionWaitRegistry({ maxWaitersPerSession: 10, maxWaitersPerOwner: 5, maxWaitersTotal: 100 }); + const held = reg.waitForSignal('s1', { until: ['exit'], timeoutMs: 5000, owner: 'alice' }); + const controller = new AbortController(); + const racing = reg.waitForSignal('s1', { + until: ['stop'], + timeoutMs: 5000, + owner: 'alice', + abortSignal: controller.signal, + }); + reg.notifySignal('s1', 'stop'); + controller.abort(); + await racing; + // A second release would have dropped the count to 0 and handed out a slot the + // still-pending waiter is using. + expect(reg.ownerWaiterCount('alice')).toBe(1); + reg.cancelAll('s1'); + await held; + expect(reg.ownerWaiterCount('alice')).toBe(0); + }); + + it('assertCapacity is public, so a route can refuse BEFORE doing expensive work', () => { + const reg = new SessionWaitRegistry({ maxWaitersPerSession: 1, maxWaitersTotal: 100 }); + expect(() => reg.assertCapacity('s1')).not.toThrow(); + void reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + expect(() => reg.assertCapacity('s1')).toThrow(WaitCapacityError); + reg.cancelEverything(); + }); + + it('checks capacity BEFORE scanning initialText', async () => { + // from=buffer hands us up to MAX_BUFFER_SCAN_BYTES that the route produced by + // joining the whole 32MB accumulator. Paying that for a request about to be + // refused turns the cap into an amplifier instead of a protection. + const reg = new SessionWaitRegistry({ maxWaitersPerSession: 1, maxWaitersTotal: 100 }); + void reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + expect(() => + reg.waitForOutput('s1', { match: 'PRESENT', timeoutMs: 5000, initialText: 'already PRESENT here' }) + ).toThrow(WaitCapacityError); + reg.cancelEverything(); + }); + + it('reports the scope in the message for every cap', () => { + const reg = new SessionWaitRegistry({ maxWaitersPerSession: 1, maxWaitersPerOwner: 1, maxWaitersTotal: 2 }); + void reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000, owner: 'alice' }); + expect(() => reg.assertCapacity('s1', 'alice')).toThrow(/scope: owner/); + expect(() => reg.assertCapacity('s1')).toThrow(/scope: session/); + void reg.waitForSignal('s2', { until: ['stop'], timeoutMs: 5000 }); + expect(() => reg.assertCapacity('s3')).toThrow(/scope: total/); + reg.cancelEverything(); + }); +}); + +describe('SessionWaitRegistry: output waits', () => { + it('matches a literal string in a chunk', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'BUILD OK', timeoutMs: 5000 }); + expect(reg.notifyOutput('s1', 'running tests...\nBUILD OK\n')).toBe(1); + const result = await promise; + expect(result.matched).toBe(true); + expect(result.immediate).toBe(false); + expect(result.snippet).toContain('BUILD OK'); + }); + + it('strips ANSI before matching', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'BUILD OK', timeoutMs: 5000 }); + reg.notifyOutput('s1', '\x1b[32mBUILD\x1b[0m OK\n'); + expect((await promise).matched).toBe(true); + }); + + it('matches across a chunk boundary', async () => { + // The carry buffer exists for exactly this: PTY chunking is arbitrary. + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'BUILD OK', timeoutMs: 5000 }); + expect(reg.notifyOutput('s1', 'tail of a line ... BUIL')).toBe(0); + expect(reg.notifyOutput('s1', 'D OK and more')).toBe(1); + const result = await promise; + expect(result.matched).toBe(true); + expect(result.snippet).toContain('BUILD OK'); + }); + + it('matches when an ANSI escape is SPLIT across two chunks', async () => { + // A PTY read boundary lands inside an escape all the time under tmux. stripAnsi + // needs a complete sequence, so the fragment used to survive the strip, land in + // the carry and split the needle: the same output matched or not depending on + // where the kernel cut the read. + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'OK', timeoutMs: 5000 }); + expect(reg.notifyOutput('s1', 'O\x1b[')).toBe(0); + expect(reg.notifyOutput('s1', '0mK\n')).toBe(1); + const result = await promise; + expect(result.matched).toBe(true); + expect(result.snippet).toContain('OK'); + }); + + it('matches when the escape is split at every offset inside the sequence', async () => { + const sequence = '\x1b[1;32m'; + for (let cut = 1; cut < sequence.length; cut++) { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'DONE', timeoutMs: 5000 }); + reg.notifyOutput('s1', `DO${sequence.slice(0, cut)}`); + reg.notifyOutput('s1', `${sequence.slice(cut)}NE`); + expect((await promise).matched, `split after ${cut} chars`).toBe(true); + } + }); + + it('matches when an OSC sequence is split across chunks', async () => { + // tmux emits ESC ] 0 ; ESC \ on every pane-title change, and the ST + // terminator is itself an ESC, which a naive "last ESC is pending" rule mishandles. + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'READY', timeoutMs: 5000 }); + expect(reg.notifyOutput('s1', 'REA\x1b]0;some pane title')).toBe(0); + expect(reg.notifyOutput('s1', '\x1b\\DY')).toBe(1); + expect((await promise).matched).toBe(true); + }); + + it('matches when a charset-select escape is split across chunks', async () => { + // ESC ( B is only removed once its FINAL byte arrives, so a cut between the '(' and + // the 'B' smuggles an escape into the haystack: neither half is a complete sequence + // on its own, and holding back only a lone trailing ESC does not cover it. This cut + // point is the one the widened pending rule exists for. + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'MARKER', timeoutMs: 5000 }); + expect(reg.notifyOutput('s1', 'MAR\x1b(')).toBe(0); + expect(reg.notifyOutput('s1', 'BKER')).toBe(1); + expect((await promise).matched).toBe(true); + }); + + it('matches when a DCS string is split across chunks', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'ABCD', timeoutMs: 5000 }); + expect(reg.notifyOutput('s1', 'AB\x1bPsome')).toBe(0); + expect(reg.notifyOutput('s1', '-dcs\x1b\\CD')).toBe(1); + expect((await promise).matched).toBe(true); + }); + + it('releases a held-back fragment that turns out not to be an escape', async () => { + // A lone trailing ESC is withheld; if the next chunk shows it was never the start of + // a sequence we remove, the text after it must still reach the haystack rather than + // being withheld behind it. + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'HELLO', timeoutMs: 5000 }); + expect(reg.notifyOutput('s1', 'X\x1b')).toBe(0); + expect(reg.notifyOutput('s1', '\nHELLO')).toBe(1); + expect((await promise).matched).toBe(true); + }); + + it('does not withhold an unterminated escape forever', async () => { + // An OSC that never terminates would otherwise stall every match on the session. + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'MARKER', timeoutMs: 5000 }); + reg.notifyOutput('s1', `\x1b]0;${'x'.repeat(600)}`); + reg.notifyOutput('s1', 'MARKER'); + expect((await promise).matched).toBe(true); + }); + + it('drops the held-back fragment when the last waiter goes away', async () => { + const reg = new SessionWaitRegistry(); + const first = reg.waitForOutput('s1', { match: 'never', timeoutMs: 20 }); + reg.notifyOutput('s1', 'text\x1b['); + await first; + // Nothing per-session may outlive the waiter set. A leaked '\x1b[' would prefix + // the next chunk and be stripped away together with the '0m' the needle wants. + const next = reg.waitForOutput('s1', { match: '0mFOUND', timeoutMs: 5000 }); + expect(reg.notifyOutput('s1', '0mFOUND')).toBe(1); + expect((await next).matched).toBe(true); + }); + + it('does not re-match text already scanned in a previous chunk', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'zzz', timeoutMs: 30 }); + reg.notifyOutput('s1', 'aaa bbb ccc'); + reg.notifyOutput('s1', 'ddd eee fff'); + expect((await promise).timedOut).toBe(true); + }); + + it('honors nocase but reports the original-case text', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'build ok', nocase: true, timeoutMs: 5000 }); + reg.notifyOutput('s1', '>>> BUILD OK <<<'); + const result = await promise; + expect(result.matched).toBe(true); + expect(result.snippet).toContain('BUILD OK'); + }); + + it('is case-sensitive by default', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'build ok', timeoutMs: 30 }); + reg.notifyOutput('s1', 'BUILD OK'); + expect((await promise).timedOut).toBe(true); + }); + + it('keeps the nocase snippet on the match when lowercasing changes length', async () => { + // 'İ' (U+0130) lowercases to TWO code units, so the index found in the lowercased + // haystack is not an index into the original. With enough of them ahead of the + // match the window slid clean off it and the snippet came back empty. + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'done', nocase: true, timeoutMs: 5000 }); + reg.notifyOutput('s1', `${'\u0130'.repeat(200)}${'x'.repeat(300)} DONE marker`); + const result = await promise; + expect(result.matched).toBe(true); + expect(result.snippet).toContain('DONE marker'); + }); + + it('reports original-case context around a length-shifted match', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'result', nocase: true, timeoutMs: 5000 }); + reg.notifyOutput('s1', `${'\u0130'.repeat(40)}before RESULT after`); + const snippet = (await promise).snippet ?? ''; + expect(snippet).toContain('before RESULT after'); + }); + + it('carries enough text for a nocase needle that lowercases longer than it is', async () => { + // needleCmp, not needle, sizes the carry: a needle of 'İ' compares as two code + // units, so two code units of pane output satisfy one needle character and a carry + // measured from the original needle drops the straddle. + const reg = new SessionWaitRegistry(); + const needle = '\u0130'.repeat(120); + const printed = 'i\u0307'.repeat(120); + const promise = reg.waitForOutput('s1', { match: needle, nocase: true, timeoutMs: 5000 }); + expect(reg.notifyOutput('s1', printed.slice(0, printed.length - 1))).toBe(0); + expect(reg.notifyOutput('s1', printed.slice(printed.length - 1))).toBe(1); + expect((await promise).matched).toBe(true); + }); + + it('scans initialText first and resolves immediately (from=buffer)', async () => { + const reg = new SessionWaitRegistry(); + const result = await reg.waitForOutput('s1', { + match: 'BUILD OK', + timeoutMs: 5000, + initialText: 'earlier output\n\x1b[32mBUILD OK\x1b[0m\n', + }); + expect(result.matched).toBe(true); + expect(result.immediate).toBe(true); + expect(result.waitedMs).toBe(0); + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('carries the tail of initialText so a match can straddle buffer and stream', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { + match: 'BUILD OK', + timeoutMs: 5000, + initialText: 'stuff that ends with BUIL', + }); + expect(reg.notifyOutput('s1', 'D OK')).toBe(1); + expect((await promise).matched).toBe(true); + }); + + it('collapses blank runs in the snippet so pane padding does not eat the context', async () => { + // A real pane pads with dozens of \r\n between the prompt and the output; without + // collapsing, the whole 80-char context window is newlines. + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'MARKER', timeoutMs: 5000 }); + reg.notifyOutput('s1', `prompt$${'\n\r'.repeat(30)}echo MARKER`); + const snippet = (await promise).snippet ?? ''; + expect(snippet).toContain('echo MARKER'); + expect(snippet).not.toMatch(/[\r\n]{2}/); + // Context on the far side of the padding survives the collapse. + expect(snippet).toContain('prompt$'); + }); + + it('matches across the charset-select escape a real bash prompt emits', async () => { + // The E2E transcript: a stock prompt renders `arkon@tnode:~/dir$` and emits + // `arkon@tnode\x1b(B\x1b[m:`. stripAnsi removes the CSI and leaves ESC ( B, so + // `match=tnode:` failed on a prompt that plainly reads `tnode:` while `match=(B` + // succeeded. Silent: a full-length timeout, no error anywhere. + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'tnode:~/codeman-cases', timeoutMs: 5000 }); + reg.notifyOutput('s1', 'clear\narkon@tnode\x1b(B\x1b[m:~/codeman-cases/e2edocs-shell$ '); + expect((await promise).matched).toBe(true); + }); + + it('leaves no charset residue in the snippet either', async () => { + // The same defect seen from the other side: the ESC was removed but the literal + // "(B" stayed as visible text in the snippet handed to the agent. + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'MARKER', timeoutMs: 5000 }); + reg.notifyOutput('s1', 'arkon@tnode\x1b(B:/tmp/e2e\x1b(B$ MARKER done'); + const snippet = (await promise).snippet ?? ''; + expect(snippet).toContain('arkon@tnode:/tmp/e2e$ MARKER done'); + expect(snippet).not.toContain('(B'); + }); + + it('matches across the other escape families stripAnsi does not know', async () => { + const cases: Array<[string, string]> = [ + ['ESC c full reset', 'AB\x1bcCD'], + ['ESC 7 / ESC 8 cursor save', 'AB\x1b7CD\x1b8'], + ['ESC ( 0 line drawing', 'AB\x1b(0CD'], + ['ESC # 8 DEC alignment', 'AB\x1b#8CD'], + ['DCS string', 'AB\x1bPsome-dcs\x1b\\CD'], + ['APC string', 'AB\x1b_Gfile=1\x1b\\CD'], + ['PM string', 'AB\x1b^private\x1b\\CD'], + ]; + for (const [label, printed] of cases) { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'ABCD', timeoutMs: 5000 }); + reg.notifyOutput('s1', printed); + expect((await promise).matched, label).toBe(true); + } + }); + + it('does not delete real output after an UNTERMINATED string sequence', async () => { + // The terminator is required on purpose: a greedy match would swallow every byte to + // the end of the window, which is the output an agent is waiting for. + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'IMPORTANT', timeoutMs: 5000 }); + reg.notifyOutput('s1', `\x1bP${'x'.repeat(600)} IMPORTANT\n`); + expect((await promise).matched).toBe(true); + }); + + it('leaves ordinary text alone', async () => { + // The residue pass is anchored on ESC, so nothing without one can be eaten. + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'a(B)c [0m ]0; #8 P^_', timeoutMs: 5000 }); + reg.notifyOutput('s1', 'literal a(B)c [0m ]0; #8 P^_ text'); + expect((await promise).matched).toBe(true); + }); + + it('never ships raw control bytes in the snippet', async () => { + // stripAnsi handles three escape families and nothing else, so ESC ( B (real, seen + // in a live bash pane) and ESC c (a full terminal reset) used to reach the calling + // agent's own terminal through `jq -r .data.snippet`. + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'MARKER', timeoutMs: 5000 }); + reg.notifyOutput('s1', 'user@host\x1b(B:/tmp\x1bc \x07\x00MARKER\x1b(0qq done\n'); + const snippet = (await promise).snippet ?? ''; + expect(snippet).toContain('MARKER'); + expect(snippet).toContain('user@host'); + // eslint-disable-next-line no-control-regex + expect(snippet).not.toMatch(/[\x00-\x08\x0b\x0c\x0e-\x1f\x7f-\x9f]/); + }); + + it('keeps ordinary whitespace in the snippet', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'MARKER', timeoutMs: 5000 }); + reg.notifyOutput('s1', 'col1\tcol2\nMARKER here'); + const snippet = (await promise).snippet ?? ''; + expect(snippet).toContain('col1\tcol2'); + expect(snippet).toContain('\n'); + }); + + it('bounds the snippet around the match', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'NEEDLE', timeoutMs: 5000 }); + reg.notifyOutput('s1', `${'x'.repeat(500)}NEEDLE${'y'.repeat(500)}`); + const snippet = (await promise).snippet ?? ''; + expect(snippet).toContain('NEEDLE'); + expect(snippet.length).toBeLessThan(300); + }); + + it('times out without erroring', async () => { + const reg = new SessionWaitRegistry(); + const result = await reg.waitForOutput('s1', { match: 'never', timeoutMs: 20 }); + expect(result.timedOut).toBe(true); + expect(result.matched).toBe(false); + expect(result.snippet).toBeNull(); + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('cancelAll resolves output waiters with ended', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForOutput('s1', { match: 'never', timeoutMs: 5000 }); + expect(reg.cancelAll('s1')).toBe(1); + const result = await promise; + expect(result.ended).toBe(true); + expect(result.matched).toBe(false); + }); + + it('notifyOutput is a no-op with no waiters', () => { + const reg = new SessionWaitRegistry(); + expect(reg.notifyOutput('s1', 'anything at all')).toBe(0); + }); + + it('resolves only the waiters whose needle matched', async () => { + const reg = new SessionWaitRegistry(); + const a = reg.waitForOutput('s1', { match: 'one', timeoutMs: 5000 }); + const b = reg.waitForOutput('s1', { match: 'two', timeoutMs: 5000 }); + expect(reg.outputWaiterCount('s1')).toBe(2); + + expect(reg.notifyOutput('s1', 'one')).toBe(1); + expect((await a).matched).toBe(true); + expect(reg.outputWaiterCount('s1')).toBe(1); + + expect(reg.notifyOutput('s1', 'two')).toBe(1); + expect((await b).matched).toBe(true); + expect(reg.totalWaiterCount()).toBe(0); + }); +}); + +describe('SessionWaitRegistry: teardown and leak prevention', () => { + it('drops the per-session bookkeeping once the last waiter resolves', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 5000 }); + expect(reg.totalWaiterCount()).toBe(1); + reg.notifySignal('s1', 'stop'); + await promise; + expect(reg.signalWaiterCount('s1')).toBe(0); + expect(reg.waiterCount('s1')).toBe(0); + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('a timed-out waiter is removed, not left in the map', async () => { + const reg = new SessionWaitRegistry(); + await reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 20 }); + await reg.waitForOutput('s1', { match: 'x', timeoutMs: 20 }); + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('cancelEverything resolves every waiter across every session', async () => { + const reg = new SessionWaitRegistry(); + const promises = [ + reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 60_000 }), + reg.waitForSignal('s2', { until: ['idle'], timeoutMs: 60_000 }), + reg.waitForOutput('s2', { match: 'nope', timeoutMs: 60_000 }), + ]; + expect(reg.totalWaiterCount()).toBe(3); + expect(reg.cancelEverything()).toBe(3); + + const results = await Promise.all(promises); + for (const result of results) expect(result.ended).toBe(true); + expect(reg.totalWaiterCount()).toBe(0); + }); + + it('a resolved waiter is not re-settled when its timeout would have fired', async () => { + const reg = new SessionWaitRegistry(); + const promise = reg.waitForSignal('s1', { until: ['stop'], timeoutMs: 25 }); + reg.notifySignal('s1', 'stop'); + const result = await promise; + await sleep(60); + expect(result.signal).toBe('stop'); + expect(result.timedOut).toBe(false); + expect(reg.totalWaiterCount()).toBe(0); + }); +}); diff --git a/test/tmux-manager.test.ts b/test/tmux-manager.test.ts index 38745e21..7388cfdd 100644 --- a/test/tmux-manager.test.ts +++ b/test/tmux-manager.test.ts @@ -247,14 +247,41 @@ describe('TmuxManager (unit)', () => { }); describe('environment exports', () => { - it('keeps COLORTERM unset for OpenCode sessions', () => { - const exports = ( + const callBuildEnvExports = (mode: string) => + ( manager as unknown as { buildEnvExports(sessionId: string, muxName: string, mode: string): string[]; } - ).buildEnvExports('session-1', 'codeman-abc12345', 'opencode'); + ).buildEnvExports('session-1', 'codeman-abc12345', mode); - expect(exports).toContain('unset COLORTERM'); + it('keeps COLORTERM unset for OpenCode sessions', () => { + expect(callBuildEnvExports('opencode')).toContain('unset COLORTERM'); + }); + + it('exports the server-stamped CODEMAN_API_URL verbatim', () => { + const original = process.env.CODEMAN_API_URL; + process.env.CODEMAN_API_URL = 'https://127.0.0.1:3199'; + try { + expect(callBuildEnvExports('claude')).toContain('export CODEMAN_API_URL=https://127.0.0.1:3199'); + } finally { + if (original === undefined) delete process.env.CODEMAN_API_URL; + else process.env.CODEMAN_API_URL = original; + } + }); + + // A hardcoded fallback exported the wrong scheme on HTTPS installs; unset must + // stay unset so in-session guards fail closed instead of curling a bad URL. + it('exports no CODEMAN_API_URL at all when the server has not stamped one', () => { + const original = process.env.CODEMAN_API_URL; + delete process.env.CODEMAN_API_URL; + try { + const exports = callBuildEnvExports('claude'); + expect(exports.some((line) => line.startsWith('export CODEMAN_API_URL'))).toBe(false); + expect(exports.join(' ')).not.toContain('localhost:3000'); + } finally { + if (original === undefined) delete process.env.CODEMAN_API_URL; + else process.env.CODEMAN_API_URL = original; + } }); });