diff --git a/.changeset/watching-badge.md b/.changeset/watching-badge.md index b7301003..e85bb212 100644 --- a/.changeset/watching-badge.md +++ b/.changeset/watching-badge.md @@ -4,9 +4,11 @@ A session watching work it started itself no longer asks you to look at it. An agent that arms a monitor, backgrounds a shell or hands a task to a cloud session is told to end its turn, so the pane goes quiet, Claude Code's idle notification lands a minute later, and the session shows up under NEEDS YOU with nothing for anyone to answer. -A CLI states what it is still running on its own screen, and the CLI registry now carries that row as an optional `capabilities.workDetect.watchingLine`. The idle probe reads the label off the capture it already takes when a turn ends, and it reaches the session payload as `watching`. Claude writes its chip on the last row (`· 1 monitor ·`, also shells, cloud sessions and background tasks); Codex pins `1 background terminal running · /ps to view` above its composer and declares the deeper window that needs. Both patterns were measured against live panes, and a CLI that declares none reports no background work, as every CLI did before. +A CLI states what it is still running on its own screen, and the CLI registry now carries that row as an optional `capabilities.workDetect.watchingLine`. The idle probe reads the label off the capture it already takes when a turn ends, and it reaches the session payload as `watching`. Claude writes its chip on the last row (`· 1 monitor ·`, also shells, cloud sessions and background tasks); Codex pins `1 background terminal running · /ps to view · /stop to close` above its composer and declares the deeper window that needs. Both patterns were measured against live panes, and a CLI that declares none reports no background work, as every CLI did before. The label is read from the foot of the screen only, because it comes off the agent's own pane: `docs/cli-registry.md` sets out what a new pattern has to satisfy before it may be trusted to quiet an alert. -The prompt that follows then opens already acknowledged. It stays in the Approvals drawer, stays answerable and stays Read My Mind context, and only the alert it would have armed is spent: no tab alert, no desktop notification, no push, and no NEEDS YOU row on any surface, `codeman tui` included. The card says why, reading "quiet, watching 1 monitor" rather than implying somebody looked. It re-arms by itself, because the next idle prompt supersedes this item and is built fresh. A permission or question dialog still goes red whatever else the agent started. +The prompt that follows then opens already acknowledged. It stays in the Approvals drawer, stays answerable and stays Read My Mind context, and only the alert it would have armed is spent: no tab alert, no desktop notification, no push, and no NEEDS YOU row on any surface, `codeman tui` included. The card says why on both surfaces, reading "quiet, watching 1 monitor" rather than implying somebody looked. It re-arms by itself, because the next idle prompt supersedes this item and is built fresh. A permission prompt or a question dialog still goes red whatever else the agent started. + +One limit worth knowing: a question asked in plain prose is not a dialog, so an agent that starts a monitor and then writes "which branch should I target?" goes quiet along with the false alarms until the background work ends. `docs/wiki/Notifications-And-Approvals.md` says so where users will meet it. Sessions also wear a `watching` badge beside their state pill on the phone overview, the desktop home rail and the rich sidebar and rail rows. diff --git a/CLAUDE.md b/CLAUDE.md index b0ae59d5..f3686f77 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -201,7 +201,7 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph **Idle detection**: Multi-layer (completion message → AI check → output silence → token stability). See `docs/respawn-state-machine.md`. -⚠️ **A `❯` sighting is NOT the end of a turn, and neither is silence.** Claude redraws the composer (`❯`) about once a second all through a turn, so the old "saw a ❯, wait 2s → idle" rule flipped every working session to idle two seconds in (measured: a session mid-tool-call at 17 minutes reporting `status:"idle"`). Its working indicator is `✻ Actualizing… (13m 23s · ↓ 47.5k tokens)`: the glyph animates through `· ✢ ✳ ∗ ✻ ✽`, the gerund is randomized, and the finished line (`✻ Cooked for 2m 49s`) carries the same glyph, so neither `SPINNER_PATTERN` (braille, not what current versions draw) nor a keyword list can see it. Matching the new line in the STREAM does not work either: tmux ships partial repaints, so the whole line reaches the PTY only every few tens of seconds. So: `_confirmIdle()` (session.ts) requires the pane to go quiet, and then asks the SCREEN via `capturePaneText()` + `CLAUDE_WORKING_LINE_PATTERN` before believing it; a sustained run of repaints (`session-activity.ts`, pure + unit tested) is what marks a turn as started, with the same screen probe vetoing keystroke echo. Idle now lands ~3-5s after a turn ends instead of 2s into one. ⚠️ **The composer glyph and the working line are per-CLI registry DATA** (`capabilities.workDetect`, #385), not Claude constants: claude declares `❯` plus the pattern above, codex declares `›` plus `[Ee]sc to interrupt`, and a CLI that declares neither falls back to Claude's pair, which is what every session used before the registry carried one. Before that, this whole mechanism was gated Claude-mode-only on the reasoning that an external CLI has no `❯`, which was true and still left every Codex session reporting `idle` for its entire life. ⚠️ `workingLine` is config-supplied (a user `clis.json` can set it) and the compiled pattern runs on the PTY hot path, so it goes through `compileVersionRegex()` in BOTH the schema refine and `_workingLinePattern()`: a nested quantifier there is a ReDoS against the event loop, and the helper returns null rather than throwing so the fallback is structural. ⚠️ **A quiet pane is not always a pane that wants you.** An agent that arms a monitor, backgrounds a shell or hands work to a cloud session is told to end its turn, so the pane goes quiet, Claude Code's `idle_prompt` notification lands a minute later and every surface files the session under NEEDS YOU with nothing to answer. The CLI states what it is still running on the last row of its screen (`⏵⏵ bypass permissions on · 1 monitor · ← for agents`), which is the optional `capabilities.workDetect.watchingLine`: `_confirmIdle()`'s own capture feeds `watchingLabel()` (`session-activity.ts`, pure), and the label lands on `Session.watching`. ⚠️ **The fix is the alert that does not fire, not the badge.** `hook-event-routes` passes that label to `notePrompt()`, which opens the idle item ALREADY acknowledged (`acknowledgedAt` + `acknowledgedReason`), so the prompt stays pending, answerable and Read-My-Mind context while the alert it would have armed is spent — the same `acknowledge()` semantics a human viewing the session has always produced. The broadcast carries `acknowledgedReason` so a live page declines to arm (`_onHookIdlePrompt`), the push is skipped, a reloading page reads `acknowledgedAt` in `seedApprovals()`, and `classifySession()` ignores an acknowledged item so `codeman tui` agrees. It re-arms for free: the next idle prompt supersedes this item and is built fresh. ⚠️ Only `idle` is eligible, so a permission or question dialog still goes red whatever else the agent started. The badge is the cosmetic half, and it rides BESIDE the state pill on the surfaces that have one, never in place of it. ⚠️ The label is **pane-derived and therefore prompt-injectable**: the search is confined to the last few non-blank rows (`WATCHING_TAIL_LINES`, or the CLI's own `watchingLines` — claude writes its chip on the last row, codex pins one above its composer) and each pattern anchors on chrome only that CLI draws, because an agent that got a bare `1 monitor` matched would silence its own alert by printing it. Tests: `test/session-watching.test.ts` and `test/watching-no-alert.test.ts` (the negative one). +⚠️ **A `❯` sighting is NOT the end of a turn, and neither is silence.** Claude redraws the composer (`❯`) about once a second all through a turn, so the old "saw a ❯, wait 2s → idle" rule flipped every working session to idle two seconds in (measured: a session mid-tool-call at 17 minutes reporting `status:"idle"`). Its working indicator is `✻ Actualizing… (13m 23s · ↓ 47.5k tokens)`: the glyph animates through `· ✢ ✳ ∗ ✻ ✽`, the gerund is randomized, and the finished line (`✻ Cooked for 2m 49s`) carries the same glyph, so neither `SPINNER_PATTERN` (braille, not what current versions draw) nor a keyword list can see it. Matching the new line in the STREAM does not work either: tmux ships partial repaints, so the whole line reaches the PTY only every few tens of seconds. So: `_confirmIdle()` (session.ts) requires the pane to go quiet, and then asks the SCREEN via `capturePaneText()` + `CLAUDE_WORKING_LINE_PATTERN` before believing it; a sustained run of repaints (`session-activity.ts`, pure + unit tested) is what marks a turn as started, with the same screen probe vetoing keystroke echo. Idle now lands ~3-5s after a turn ends instead of 2s into one. ⚠️ **The composer glyph and the working line are per-CLI registry DATA** (`capabilities.workDetect`, #385), not Claude constants: claude declares `❯` plus the pattern above, codex declares `›` plus `[Ee]sc to interrupt`, and a CLI that declares neither falls back to Claude's pair, which is what every session used before the registry carried one. Before that, this whole mechanism was gated Claude-mode-only on the reasoning that an external CLI has no `❯`, which was true and still left every Codex session reporting `idle` for its entire life. ⚠️ `workingLine` is config-supplied (a user `clis.json` can set it) and the compiled pattern runs on the PTY hot path, so it goes through `compileVersionRegex()` in BOTH the schema refine and `_workingLinePattern()`: a nested quantifier there is a ReDoS against the event loop, and the helper returns null rather than throwing so the fallback is structural. ⚠️ **A quiet pane is not always a pane that wants you.** An agent that arms a monitor, backgrounds a shell or hands work to a cloud session is told to end its turn, so Claude Code's `idle_prompt` notification lands a minute later on a session that wants nothing. The CLI states what it is still running on its own screen (optional `capabilities.workDetect.watchingLine`, with `watchingLines` for how far up the screen it sits), the idle probe reads it into `Session.watching`, and `notePrompt()` then opens that idle item ALREADY acknowledged so no surface alerts. ⚠️ **The fix is the alert that does not fire, not the `watching` badge**, and only `idle` is eligible — but a prose question is not a dialog, so a question asked in plain text while background work runs is silenced with the false alarms. ⚠️ **The label is pane-derived and therefore prompt-injectable**: the window must cover only rows the CLI draws and each pattern must anchor on chrome only that CLI can produce, or an agent silences its own alert by printing the words. → [architecture-invariants#the-watching-signal-a-quiet-pane-that-is-not-waiting-for-you](docs/architecture-invariants.md#the-watching-signal-a-quiet-pane-that-is-not-waiting-for-you). Tests: `test/session-watching.test.ts`, `test/watching-no-alert.test.ts`. **Workspace-trust dialog auto-accept** (`session-trust-dialog.ts`, pure + unit tested): Claude Code asks once per directory ("Is this a project you created or one you trust?") before it will read or edit anything, and since Codeman sessions run permission-skipping or classifier-guarded modes the answer is always yes, so a session parked on that dialog is simply stuck. ⚠️ **Match the compacted SCREEN, never the stream.** tmux repaints a row by writing each word and then a cursor-forward (`\x1b[C`) instead of a space, and Ink colours each word separately, so the wire carries `I\x1b[Ctrust\x1b[Cthis\x1b[Cfolder`; stripping the escapes leaves `Itrustthisfolder`, because the spaces are not there to strip, they were never sent. A plain `includes('trust this folder')` therefore never matched a single chunk and the auto-accept was silently DEAD for every session that hit the dialog. `compactScreenText()` removes ALL whitespace instead (plus the `ESC ( B` charset selects that `stripAnsi` does not cover, which would otherwise land inside a phrase as a literal `(B`), which survives both that repaint style and the spaced full-screen redraw. ⚠️ **Never answer it with a blind `\r`.** The layout has changed under us at least twice, and Claude Code 2.1.252 dropped the option numbers, put "No, exit" FIRST and highlights IT by default, so the Enter that answered the old dialog now picks *exit* and the pane dies (`Pane is dead (status 1)`) seconds after the session starts. `trustDialogNextKey()` reads the `❯` marker and returns ONE step at a time (an arrow while the cursor is on the wrong option, Enter only once the screen shows it on the trust option), with the pane re-read between steps, so a dropped arrow costs a repaint instead of the session; a frame that does not say which option is highlighted returns null and waits for the next repaint. ⚠️ The LAST marked option in the text wins, because the direct-PTY fallback reads an append-only buffer where every repaint since launch is still present and an older frame must not out-vote the freshest one. ⚠️ Answering types into a live session, so THREE guards must all hold and none is redundant: a **startup-only window** (`TRUST_DIALOG_WINDOW_MS`, 90s, since the dialog renders before the main UI and leaving it open forever would let an agent transcript that merely QUOTES the dialog trigger an Enter, this file being an example), a **two-marker match** requiring a trust phrase AND one of the dialog's own confirm affordances (`isTrustDialogScreen`), and an **attempt cap** (`TRUST_DIALOG_MAX_ATTEMPTS`, 6: a keystroke can land while Ink is still mounting the widget and be dropped, which is the other half of why sessions got stuck here, but retrying forever would hammer keys into whatever came next; it was 3 while one Enter answered the dialog, and answering now costs at least two keystrokes). ⚠️ It reads `capturePaneText()` and falls back to a deliberately SHORT tail of the terminal buffer only on a direct-PTY session, which has no pane: that buffer is append-only, so a longer tail would keep re-matching a dialog answered minutes ago. ⚠️ **The scan must schedule its own next read** (`_trustDialogTimer`, cleared in `_clearAllTimers()`): it runs from the PTY `onData` handler, which was enough while one Enter answered the dialog, but the arrow that moves the cursor is the LAST output the pane produces, so a two-keystroke answer waiting on more output parks forever with the cursor sitting on the right option (measured on a live 2.1.252 spawn: cursor moved at 6 s, then nothing). diff --git a/docs/architecture-invariants.md b/docs/architecture-invariants.md index b024875a..bd1e6b9a 100644 --- a/docs/architecture-invariants.md +++ b/docs/architecture-invariants.md @@ -90,6 +90,14 @@ Tests: `test/docker-hosts.test.ts`, `test/docker-exec-options.test.ts`, `test/do **Signal availability is decided by MODE, not by `isExternalCliMode()`.** `stop` and `blocked` come from Claude Code hooks, so `hooksAvailableForMode()` is true only for `claude`: `shell` is not an external-CLI mode but installs no hooks either, so `until=stop` on a shell session is a guaranteed unresolvable wait dressed up as a timeout. The behavior split is deliberate and must survive refactors: an EXPLICIT request for an unavailable signal is a 400 naming the mode, while the DEFAULT set (`stop,idle,exit`) silently drops them and echoes the narrowed set back as `wait.until`, because omitting the parameter must never 400. Signal quality is not uniform either: `stop` is definitive (Claude Code says the turn is over), `idle` is inferred from output stabilization plus prompt detection and can flap mid-turn when a spinner pauses, which is why `stop` is the documented default to orchestrate on and `idle` is the fallback for sessions that emit no hooks. ⚠️ **`idle` being ACCEPTED for a mode does not mean it ever FIRES there.** `startShell()` emits exactly one `idle` on a 500ms readiness timer and nothing afterwards, so a shell session sits at `status:'idle'` no matter what its pane is doing; since send-and-wait and `fresh=1` both require a TRANSITION, both can only time out on a shell worker (measured: a default `wait` on `sleep 4` burned its full 25s). The ❯-prompt and spinner detection that drives the real `working`/`idle` cycle is Claude's output format, so hook-less modes synchronize with `wait-output` markers or `exit`, and the docs must say so rather than listing `idle` as "available" and letting the reader infer it is usable. ⚠️ **Signals are edge-triggered with no history**, and that is a real orchestration limit: a signal that fires while no waiter is registered is gone, unobservable by any later wait variant (`until=stop` after the turn ended just times out, `fresh` or not — measured, R2-A). Documented client patterns must therefore register the waiter before the event can fire (send-and-wait) or gather on latched `wait-output` markers with `from=buffer`; the skill's fan-out flow was rewritten accordingly, and "fire-and-forget N prompts, then gather signal-waits sequentially" must never be documented again. The durable fix, a server-side latched last-signal-per-turn, is deferred with Part 3 of `docs/agent-control-plan.md`. ⚠️ Relatedly, the route corrects liveness that `SessionStatus` cannot express: `currentSignalFor()` answers `exit` whenever `pid === null` (exited, detached, or created-and-never-started) **or the mux pane is dead**, because `Session` parks a DEAD PTY at `_status = 'idle'` and trusting the status would answer the default wait `{signal:'idle', immediate:true}` for a crashed worker while `until=exit` blocked forever on an event that already happened. ⚠️ **`pid` alone cannot carry liveness for a tmux-backed session**: that pid is the local `tmux attach` CLIENT, so a worker exiting inside its pane leaves `pane_dead=1` with the client alive and `pid` never goes null — the `pid === null` branch is unreachable in the normal configuration (unit tests exercise it because `MockSession` sets `pid` by hand; only a live instance showed the gap). Liveness is therefore probed at the mux layer: `workerIsDead()` consults `mux.isPaneDead(muxName)` with a ~750 ms per-pane cache, ONLY on blocking waits (measured: 0 tmux execs across 100 non-wait input POSTs — the browser hot path pays nothing), plus a refcounted 3 s `watchForDeadWorker` interval so a worker dying while a wait is parked resolves it in ~3 s instead of burning the timeout. The probe fails SAFE by construction ("cannot tell" is never "dead": non-mux sessions, a missing or throwing `isPaneDead`, all return false). On send-and-wait, a "successful" `send-keys` into a dead pane additionally overrides `delivered` to `false` and rolls the dedup seq back (`undoOnFailure`), because the bytes went nowhere and a retry against a restarted worker must not be refused as a duplicate. The cost is that a just-created session reads as `exit`, which the wire docs must spell out as "not started yet"; the fix lives at the route rather than in `signalForStatus()` because the registry deliberately holds no `Session` reference. ⚠️ Two `claude`-mode cases still lose hooks for reasons outside the registry: a Docker case cannot reach a loopback-bound Codeman without `CODEMAN_DOCKER_BRIDGE_HOOKS=1`, and a remote-SSH case runs the agent on another host whose hooks may never reach this server. Bounds are env-overridable (`CODEMAN_WAIT_MAX_MS`, `CODEMAN_WAIT_DEFAULT_MS`, `CODEMAN_WAIT_MAX_PER_SESSION`, `CODEMAN_WAIT_MAX_PER_OWNER`, `CODEMAN_WAIT_MAX_TOTAL`, `CODEMAN_WAIT_BUFFER_SCAN_BYTES`) and each is clamped to a hard bound so a typo degrades to the default instead of disabling the protection; they are internal tuning knobs like the rest of `src/config/`, NOT part of the SemVer-covered env-var surface in `versioning-policy.md`. Tests: `test/session-wait-registry.test.ts`, `test/routes/session-wait-routes.test.ts`, `test/routes/session-wait-output-routes.test.ts`, `test/routes/session-input-wait.test.ts`. +### The watching signal (a quiet pane that is not waiting for you) + +**A session that armed a monitor, backgrounded a shell or started a background terminal ends its turn and goes quiet, and a minute later Claude Code's idle notification arrives.** Before this existed, that prompt became an approval item like any other, so every surface filed the session under NEEDS YOU with nothing for a human to answer. The CLI says which kind of quiet it is on its own screen, and reading that row is the whole mechanism: `capabilities.workDetect.watchingLine` (optional, per CLI) plus `watchingLines` (how many non-blank rows at the foot of the screen may hold it, default `WATCHING_TAIL_LINES` = 1). `_confirmIdle()` already captures the pane at the moment a turn ends, so `_readWatching()` runs `watchingLabel()` (pure, `session-activity.ts`) over that same capture; the label lands on `Session.watching` and rides `toLightDetailedState()` out to every payload. ⚠️ It is cached BESIDE `_lastPaneProbeWorking` and goes stale with it, because the probe returns its cached boolean without re-capturing inside `PANE_PROBE_MIN_INTERVAL_MS` and a label from a capture nobody took is a guess. ⚠️ It then FREEZES once `_confirmIdle()` concludes — nothing looks at the pane again until it produces output — which is correct rather than tolerable, since the background work ending is itself what wakes the agent; a timer to keep it fresh would spend a `capture-pane` per idle session per tick to learn nothing. + +**The fix is the alert that does not fire; the badge is cosmetic.** `hook-event-routes` passes the label to `notePrompt()`, which opens the idle item ALREADY acknowledged (`acknowledgedAt` + `acknowledgedReason`). Nothing new suppresses anything: `acknowledge()` has always meant "the alert this prompt armed is spent", so the item stays pending, answerable and available as Read My Mind context, and a wrong label costs a card that does not blink rather than an alert that was never created. Every surface follows from that one flag — the broadcast carries `acknowledgedReason` so a live page declines to arm (`_onHookIdlePrompt`, settings-ui.js), the push is skipped, a reloading page reads `acknowledgedAt` in `seedApprovals()` as it always did, `classifySession()` and `pendingApprovalCount()` (tui-model.ts, tui-render.ts) ignore an acknowledged item, and the TUI card drops to the `info` tone and says why. It re-arms for free: the next idle prompt supersedes the item and is built fresh. ⚠️ Only `idle` is eligible, so a permission or question dialog still goes red whatever else the agent started — but a prose question is NOT a dialog, so an agent that arms a monitor and then asks "which branch?" in plain text is silenced along with the false alarms. That is the accepted cost of the design and the reason the kind gate sits at the single place items are created. + +**⚠️ The label is pane-derived, so the window and the anchor are a trust boundary, not formatting.** An agent that gets its own text matched silences its own alert. Two things prevent it, and BOTH belong to whoever adds a pattern for a new CLI: the window must cover only rows that CLI draws, and the pattern must anchor on chrome only that CLI can produce. Claude satisfies both — its chip is the LAST row, so the default window of one row excludes even the status line directly above it, whose content comes from a `statusLine` command a bypassed session can write into its own `.claude/settings.json`. Codex does not: its row sits above the composer, and the slot it occupies holds the last row of the TRANSCRIPT whenever no terminal is running, so matching the complete row raises the bar without closing it. What contains that is `hooks: 'none'` — no hook event from a codex session reaches `notePrompt()`, so a forged label costs a wrong badge and nothing else, and a CLI that gains hook signals must not keep a pattern that soft. The label is also ANSI-stripped and capped (`MAX_WATCHING_LABEL_CHARS`) at the source, and every interpolation of it into markup goes through `escapeHtml()`, since a config-supplied capture group decides what it holds. Tests: `test/session-watching.test.ts` (the label and both CLI patterns), `test/watching-no-alert.test.ts` (the negative claim across all four surfaces). + ### Auto-resume on usage limit **Auto-resume on usage limit** ("token pause" control, opt-in per session, top of the Respawn tab): when Claude halts on a subscription limit ("5-hour limit reached ∙ resets 8pm" and all 1.0.x–2.1.x variants), `usage-limit-patterns.ts` (pure, unit-tested) parses the reset time from cleaned output; `SessionAutoOps` arms a timer for reset+2min, then sends Esc (dismisses the rate-limit dialog) + `continue`. Still-limited responses re-arm the loop (5-min retry on stale times); a `working` transition cancels it. Claude-mode only (detection rides `_processExpensiveParsers`). Persists/recovers via `SessionState.autoResumeEnabled`/`autoResumeAt`; respawn cycles are blocked while paused (`isLimitPaused` guard in `onIdleDetected` — prevents `/clear` from wiping the paused conversation). Endpoint: `POST /api/sessions/:id/auto-resume`; SSE: `session:limitPauseScheduled`/`limitResume`/`limitResumeCancelled`. Tests: `test/usage-limit-patterns.test.ts`, `test/session-auto-resume.test.ts`. diff --git a/docs/cli-registry.md b/docs/cli-registry.md index 9e343ed2..dbeb24c6 100644 --- a/docs/cli-registry.md +++ b/docs/cli-registry.md @@ -58,23 +58,33 @@ so a pane waiting for its own background work never raises an alert a human cann Group 1 is the label, and a CLI that declares no pattern reports no background work. Two CLIs declare such a row today, and they put it in different places. Claude writes its -chip on the last row of the screen, so it keeps the default window of `WATCHING_TAIL_LINES` -rows and anchors on the `·` its footer joins items with. Codex pins +chip on the last row of the screen, so it keeps the default one-row window and anchors on +the `·` its footer joins items with. Codex pins `1 background terminal running · /ps to view · /stop to close` ABOVE its composer, which puts the row third from the bottom once the status line and the composer are counted, so its -entry declares `watchingLines: 4` and anchors on the ` · /ps to view` tail. Both were -measured against live panes rather than read out of a binary, which is the standard for -adding a third. +entry declares `watchingLines: 3` and matches that row end to end. Both were measured +against live panes rather than read out of a binary, which is the standard for adding a +third. That label is the one value in the registry that an AGENT can influence, because it comes off the agent's own screen. Two things keep it honest, and both belong to whoever adds a pattern for a new CLI. `watchingLabel()` in `session-activity.ts` searches only the last few -non-blank rows, which is the part of the screen the CLI draws rather than the agent, and the -pattern anchors on chrome only that CLI can produce. Without both, an agent could silence its -own idle alert by printing the words into its output. Keep the window as small as the layout -allows, since every row it adds is another row the agent may be able to write. The label is -also ANSI-stripped and length-capped at the source, since it ends up on a badge and in an -approval card. +non-blank rows, which should be the part of the screen the CLI draws rather than the agent, +and the pattern should anchor on chrome only that CLI can produce. Keep the window as small +as the layout allows, since every row it adds is another row the agent may be able to write. +The label is also ANSI-stripped and length-capped at the source, and every interpolation of +it into markup goes through `escapeHtml()`, since it ends up on a badge and in an approval +card. + +The two shipped entries do not sit equally well behind that rule, and the difference decides +what a pattern is allowed to do. Claude's chip is the last row, so its one-row window holds +nothing the agent can write — not even the status line above it, whose command a session +running with permissions bypassed can write into its own `.claude/settings.json`. Codex's row +shares its slot with the last row of the transcript whenever no terminal is running, so a +message ending in that exact line is matched. What keeps that harmless is `hooks: 'none'`: no +hook event from a codex session reaches the approvals inbox, so a forged label costs a wrong +badge and cannot silence an alert. Before giving a CLI both hook signals and a pattern, make +sure its row is one the agent cannot write. ### Three capabilities that must stay independent diff --git a/docs/wiki/Notifications-And-Approvals.md b/docs/wiki/Notifications-And-Approvals.md index c096cd50..41472d1a 100644 --- a/docs/wiki/Notifications-And-Approvals.md +++ b/docs/wiki/Notifications-And-Approvals.md @@ -103,6 +103,30 @@ locked phone and the agent continues. With the inbox off, the buttons are stripped from the notification payload entirely rather than being shown and failing. +## When a session is watching its own work + +An agent that starts a monitor, puts a shell in the background or hands a task to a cloud +session is told by its CLI to end the turn and wait to be notified. The pane then goes +quiet, and the CLI's idle notification arrives about a minute later — for a session that +wants nothing from you. + +Codeman reads what the CLI prints about its own background work and treats that prompt +differently. It raises no tab alert, no desktop notification and no push, the session stays +out of NEEDS YOU on every surface, and the row wears a blue **watching** badge instead. Hover +it, or read it on a phone through your screen reader, and it says what is running: "1 +monitor", "2 shells", "1 background terminal". + +The prompt itself is not thrown away. It sits in the Approvals drawer as an ordinary card, +still answerable, with a line reading "quiet, watching 1 monitor" where a card you had +already looked at would say nothing. The next time that session goes quiet for an ordinary +reason, it alerts you exactly as before. + +Two limits are worth knowing. A permission prompt or a question dialog still goes red +whatever else the agent started, because that one blocks it outright. A question asked in +plain prose is not a dialog, so an agent that starts a monitor and then writes "which branch +should I target?" is quiet along with the rest — check a watching session yourself if it has +been quiet longer than the work it is waiting for should take. + ## The phone overview On phones, tapping the "C" logo gives a session overview with **NEEDS YOU** first, then diff --git a/src/config/cli-registry/schema.ts b/src/config/cli-registry/schema.ts index 29565f91..52bacff2 100644 --- a/src/config/cli-registry/schema.ts +++ b/src/config/cli-registry/schema.ts @@ -323,6 +323,14 @@ const capabilitiesSchema = z watchingLines: z.number().int().min(1).max(8).optional(), }) .strict() + // A window with nothing to search is a typo, not a configuration. Refused at LOAD + // time for the same reason `privilegedParams[].param` is checked against the params + // the entry declares: the failure is otherwise silent and looks like a feature that + // simply never fires. + .refine( + (v) => v.watchingLines === undefined || v.watchingLine !== undefined, + 'watchingLines has nothing to bound without a watchingLine' + ) .optional(), model: z .object({ source: z.enum(['flag', 'claude-settings-file', 'none']), param: z.string().optional() }) diff --git a/src/config/cli-registry/stock.ts b/src/config/cli-registry/stock.ts index 8c9adda9..fb3cd87c 100644 --- a/src/config/cli-registry/stock.ts +++ b/src/config/cli-registry/stock.ts @@ -211,12 +211,15 @@ const CLAUDE: CliEntry = { // the CLI's own words for each kind of background task, and group 1 is the one // Codeman badges the session with. Verified against a live 2.1.278 pane on // 2026-09-21. - // ⚠️ The leading `·` is an anchor, not decoration. This pattern runs over the foot - // of the screen, which is the one part of it the AGENT does not write, and the - // separator is what keeps it on the footer's own item list. An agent that could get - // a bare `1 monitor` matched would silence its own idle alert by printing it. A - // footer that ever carries the chip as its only item therefore reports no watching - // rather than opening that door. See `watchingLabel()` in `session-activity.ts`. + // ⚠️ Two things keep an agent from writing its own label here, and both matter. + // The footer is the LAST row, so the default one-row window (`WATCHING_TAIL_LINES`) + // holds nothing but Ink's own chrome — in particular it leaves out the status line + // directly above, whose content comes from a `statusLine` command a bypassed + // session can write into its own `.claude/settings.json`. And the leading `·` keeps + // the match on the footer's own item list rather than on any text that happens to + // carry a count. A footer that ever drew the chip as its only item would report no + // watching rather than open that door. See `watchingLabel()` in + // `session-activity.ts`. watchingLine: String.raw`·\s*(\d+ (?:monitors?|shells?|teams?|local agents?|cloud sessions?|MCP tasks?|background tasks?|(?:background|remote) dynamic workflows?|Artifact comment monitors?))`, }, requiresMux: false, @@ -534,16 +537,24 @@ const CODEX: CliEntry = { // running: ` 1 background terminal running · /ps to view · /stop to close`. Unlike // Claude's footer chip that row sits ABOVE the composer, which puts it third from the // bottom once the status line and the composer are counted, hence `watchingLines`. - // The ` · /ps to view` tail is the anchor: it is CLI chrome, it names a slash command - // that only the CLI can offer, and without it a bare count in the transcript would do. // Measured against a live codex-cli 0.154.0 pane on 2026-09-22: the row appears when // the terminal starts, follows the composer down as the conversation grows, and is // gone after `/stop`. + // ⚠️ This entry CANNOT promise what Claude's does, and the difference is Codex's + // layout rather than its pattern. The third row from the bottom is the chip only + // while a terminal runs; with none running it is the last row of the transcript, + // which the agent writes. Matching the complete row raises the bar — an assistant + // message has to end with this exact line, to the character — but nothing here makes + // forging it impossible, so do not read the Claude comment above as applying here. + // What contains it is that codex declares `hooks: 'none'`: no hook event from a codex + // session ever reaches `notePrompt()`, so there is no idle item to pre-acknowledge + // and a forged label costs a wrong badge and nothing else. A CLI that gains hook + // signals must not keep a pattern this soft. workDetect: { promptGlyph: '›', workingLine: '[Ee]sc to interrupt', - watchingLine: String.raw`(\d+ background terminals?) running · /ps to view`, - watchingLines: 4, + watchingLine: String.raw`^\s{0,4}(\d+ background terminals?) running · /ps to view · /stop to close$`, + watchingLines: 3, }, transcript: 'codex-rollout', altScreen: 'strip-full', diff --git a/src/session-activity.ts b/src/session-activity.ts index abb08c0a..c49cf4e3 100644 --- a/src/session-activity.ts +++ b/src/session-activity.ts @@ -98,19 +98,21 @@ export function isPaneQuiet(lastActivityAt: number, now: number, silenceMs: numb * How many rows at the foot of a pane capture may hold the background-work row, for a * CLI that declares no number of its own (`capabilities.workDetect.watchingLines`). * - * Claude Code draws its chip on the LAST row of the screen. The row above it is the - * status line, which a user's own `statusLine` command writes, and two rows is what - * covers the chip wherever a trailing blank or a one-line notice pushes it up by one. - * Codex pins its row above the composer instead and declares four. + * One, because the tightest window is the right default and Claude Code needs no more: + * it draws its chip on the LAST row of the screen. Blank rows are dropped before the + * window is taken, so a trailing blank costs nothing, and a CLI that ever prints a row + * BELOW its chip loses the badge rather than gaining a hole. * - * ⚠️ The ceiling is the security boundary, not a tidiness measure. The label is - * PANE-DERIVED, and everything on that screen above the CLI's own chrome is text the - * agent wrote itself, so an agent that printed `· 1 monitor ·` into its output would - * silence its own idle alert. Keep each CLI's window as small as its layout allows, - * keep its pattern anchored on chrome only that CLI can draw, and never widen either to - * a whole-pane search. + * ⚠️ The size of this window is a trust boundary, not a tidiness measure, and the row + * it excludes first is the one that taught us so: Claude's status line sits directly + * above the footer, its content comes from a `statusLine` command, and a session running + * with permissions bypassed can write that command into `.claude/settings.json` in its + * own workspace. A window of two therefore let an agent print `· 1 monitor ·` onto a row + * of its own and silence its own idle alert. Every row added here is another row + * somebody may be able to write, so widen this only for a CLI whose layout forces it, + * and never to a whole-pane search. */ -export const WATCHING_TAIL_LINES = 2; +export const WATCHING_TAIL_LINES = 1; /** Longest label a badge will carry. A footer chip is a handful of words. */ export const MAX_WATCHING_LABEL_CHARS = 40; diff --git a/src/session.ts b/src/session.ts index 79804c61..7f1ed406 100644 --- a/src/session.ts +++ b/src/session.ts @@ -82,6 +82,7 @@ import { isSustainedActivity, isPaneQuiet, watchingLabel, + WATCHING_TAIL_LINES, IDLE_RECHECK_MS, PANE_PROBE_MIN_INTERVAL_MS, PANE_PROBE_RECHECK_MS, @@ -517,6 +518,8 @@ export class Session extends EventEmitter { private _workingLineRe: RegExp | undefined = undefined; /** Lazily compiled `capabilities.workDetect.watchingLine`. See _watchingLinePattern(). */ private _watchingLineRe: RegExp | null | undefined = undefined; + /** Resolved with the pattern above: how many rows at the foot of the screen to search. */ + private _watchingWindow = WATCHING_TAIL_LINES; private _trustDialogAccepted: boolean = false; // Stops the trust-dialog scan (answered, or given up) private _trustDialogAttempts = 0; // Keystrokes sent at the trust dialog private _lastTrustDialogScanAt = 0; // Throttle for the trust-dialog screen read @@ -2760,9 +2763,7 @@ export class Session extends EventEmitter { // Called only from the probe, and only with what a capture returned: `null` is // "the screen could not be read", which is not evidence that nothing is running. if (!pattern || paneText === null) return; - // How far up the screen this CLI's row can sit is its own business: Claude writes on - // the last row, Codex pins one above its composer. Both stay at the foot. - this._watching = watchingLabel(paneText, pattern, getCli(this.mode)?.capabilities.workDetect?.watchingLines); + this._watching = watchingLabel(paneText, pattern, this._watchingWindow); } /** @@ -2773,8 +2774,13 @@ export class Session extends EventEmitter { */ private _watchingLinePattern(): RegExp | null { if (this._watchingLineRe === undefined) { - const src = getCli(this.mode)?.capabilities.workDetect?.watchingLine; - this._watchingLineRe = src ? compileVersionRegex(src) : null; + // The pattern and the window it runs over are one decision, so they are resolved + // together: how far up the screen a CLI's row can sit is as much a property of its + // layout as the row itself. Claude writes on the last row and keeps the default, + // Codex pins one above its composer and declares more. + const detect = getCli(this.mode)?.capabilities.workDetect; + this._watchingLineRe = detect?.watchingLine ? compileVersionRegex(detect.watchingLine) : null; + this._watchingWindow = detect?.watchingLines ?? WATCHING_TAIL_LINES; } return this._watchingLineRe; } diff --git a/src/tui/tui-approvals.ts b/src/tui/tui-approvals.ts index 2cb10dea..000c6342 100644 --- a/src/tui/tui-approvals.ts +++ b/src/tui/tui-approvals.ts @@ -28,8 +28,11 @@ import type { ApprovalItem, ApprovalOption } from '../web/approval-inbox.js'; import type { TuiApprovalAnswer } from './tui-client.js'; -/** Card severity, in the same red/yellow vocabulary the web inbox uses. */ -export type TuiApprovalTone = 'err' | 'warn'; +/** + * Card severity, in the same red/yellow vocabulary the web inbox uses, plus the quiet + * third case: an item that opened acknowledged asks for nothing and reads grey. + */ +export type TuiApprovalTone = 'err' | 'warn' | 'info'; export interface TuiApprovalCard { tone: TuiApprovalTone; @@ -50,8 +53,14 @@ function clean(text: string | undefined): string { return (text ?? '').replace(/\s+/g, ' ').trim().slice(0, MAX_CARD_TEXT); } +/** + * How loud the card is. An idle prompt the inbox opened ALREADY acknowledged is not + * asking for anything — the session is watching work it started itself — so it drops to + * `info` and out of the warning vocabulary the other two share with the web inbox. + */ export function approvalTone(item: ApprovalItem): TuiApprovalTone { - return item.kind === 'idle' ? 'warn' : 'err'; + if (item.kind !== 'idle') return 'err'; + return item.acknowledgedReason ? 'info' : 'warn'; } /** @@ -65,9 +74,13 @@ export function approvalCard(item: ApprovalItem): TuiApprovalCard { const summary = clean(item.toolSummary) || clean(item.toolName); if (item.kind === 'idle') { + // Say what it is waiting for rather than asking for a reply, in the same words the + // web drawer uses for the same item. The prompt is still answerable, so the hint + // stays either way. + const quiet = clean(item.acknowledgedReason); return { - tone: 'warn', - title: message || 'waiting for your reply', + tone: approvalTone(item), + title: quiet ? `quiet, ${quiet}` : message || 'waiting for your reply', detail: [], options: [], hint: 'p to reply', diff --git a/src/tui/tui-render.ts b/src/tui/tui-render.ts index 8033104c..1893e59d 100644 --- a/src/tui/tui-render.ts +++ b/src/tui/tui-render.ts @@ -462,7 +462,8 @@ export function computeListWindow( * The pending dialog, drawn above the tail: the question, the parsed options * with their digits, and the keys that answer them. Red for a permission or * question prompt, yellow for an idle one, the same severity vocabulary the web - * inbox uses. + * inbox uses. An idle prompt that opened acknowledged carries neither: it reads + * grey with the idle glyph, because nothing about it wants the reader. */ export function renderApprovalCard( item: ApprovalItem, @@ -472,8 +473,8 @@ export function renderApprovalCard( ): string[] { const paint = painterFor(opts.color); const card = approvalCard(item); - const color = card.tone === 'err' ? SGR.red : SGR.yellow; - const glyph = card.tone === 'err' ? glyphs.blockedPermission : glyphs.waiting; + const color = card.tone === 'err' ? SGR.red : card.tone === 'warn' ? SGR.yellow : SGR.gray; + const glyph = card.tone === 'err' ? glyphs.blockedPermission : card.tone === 'warn' ? glyphs.waiting : glyphs.idle; const lines: string[] = []; const push = (text: string, style: string): void => { lines.push(padDisplay(paint(clipStyledLine(text, width), style), width)); @@ -571,10 +572,19 @@ function previewBody( // Chrome // ───────────────────────────────────────────────────────────────────────────── -/** Sessions with a prompt waiting on a human, which is what the badge counts. */ +/** + * Sessions with a prompt waiting on a human, which is what the badge counts. + * + * An ACKNOWLEDGED item is not one of them. Its alert has been spent, either by somebody + * opening the session elsewhere or because the inbox opened it that way for a session + * watching its own background work, and the row has already left NEEDS YOU by the same + * flag (`classifySession`). Counting it here would put a number in the header for a + * group the reader can see is empty. + */ export function pendingApprovalCount(model: TuiRenderModel): number { let count = 0; - for (const group of model.groups()) for (const row of group.rows) if (row.approval) count++; + for (const group of model.groups()) + for (const row of group.rows) if (row.approval && !row.approval.acknowledgedAt) count++; return count; } diff --git a/src/web/public/app.js b/src/web/public/app.js index 02b7cba1..4de9ecb5 100644 --- a/src/web/public/app.js +++ b/src/web/public/app.js @@ -4662,6 +4662,10 @@ class CodemanApp { const prev = tab.dataset.tabState; // The since ANCHOR moves without the state changing (each new turn re-stamps // lastSubmitAt), so key the compare on both. + // Unescaped on purpose, and it still matches the attribute the initial render wrote: + // that one goes through escapeHtml() because it is interpolated into markup, and the + // browser hands the decoded string back through `dataset`. `watching` is the only + // pane-derived value in this signature, which is why it is the only one escaped there. const sig = `${row.state}:${row.since ? row.since.at : 0}:${row.createdAt}:${row.watching}`; if (tab.dataset.tabMetaSig === sig) return; tab.dataset.tabMetaSig = sig; @@ -5281,7 +5285,7 @@ class CodemanApp { const richMeta = this._sidebarRichMetaHTML(richRow); const richClass = richRow ? ` tab-state-${richRow.state}` : ''; const richData = richRow - ? ` data-tab-state="${richRow.state}" data-tab-meta-sig="${richRow.state}:${richRow.since ? richRow.since.at : 0}:${richRow.createdAt}:${richRow.watching}"` + ? ` data-tab-state="${richRow.state}" data-tab-meta-sig="${richRow.state}:${richRow.since ? richRow.since.at : 0}:${richRow.createdAt}:${escapeHtml(richRow.watching)}"` : ''; const inlineSessionActions = this.shouldInlineSessionActions(); diff --git a/src/web/public/mobile-overview.js b/src/web/public/mobile-overview.js index 49b5ed1c..940a87e8 100644 --- a/src/web/public/mobile-overview.js +++ b/src/web/public/mobile-overview.js @@ -773,7 +773,12 @@ Object.assign(CodemanApp.prototype, { badge.className = base + ' ' + base + '--watching'; badge.setAttribute('data-i18n-skip', ''); badge.textContent = WATCHING_BADGE_TEXT; + // The label rides in BOTH, because a tooltip is desktop-only: a phone has no hover + // target, and a screen reader gets the one word either way. This is the surface the + // badge was built for first, so "watching" with no way to learn what would be the + // wrong place to save a line. badge.title = watchingBadgeTitle(label); + badge.setAttribute('aria-label', watchingBadgeTitle(label)); return badge; }, diff --git a/test/cli-registry-schema.test.ts b/test/cli-registry-schema.test.ts index dcf0b9f6..abd31ff3 100644 --- a/test/cli-registry-schema.test.ts +++ b/test/cli-registry-schema.test.ts @@ -148,6 +148,16 @@ describe('workDetect.workingLine is guarded like every other config regex', () = } }); + it('refuses a window with no pattern to bound', () => { + expectRejected((e) => { + (e.capabilities as Record).workDetect = { + promptGlyph: '>', + workingLine: 'working', + watchingLines: 3, + }; + }, 'a window with nothing to search is a typo whose failure is otherwise silent'); + }); + it('accepts every shipped watchingLine', () => { for (const entry of STOCK_CLIS) { const src = entry.capabilities.workDetect?.watchingLine; diff --git a/test/mobile-overview.test.ts b/test/mobile-overview.test.ts index 6c6420c8..2269b585 100644 --- a/test/mobile-overview.test.ts +++ b/test/mobile-overview.test.ts @@ -19,8 +19,11 @@ function fakeElement(): any { type: '', dataset: {}, style: {}, + attrs: {} as Record, children: [] as any[], - setAttribute() {}, + setAttribute(name: string, value: string) { + el.attrs[name] = value; + }, appendChild(child: any) { el.children.push(child); return child; @@ -514,13 +517,16 @@ describe('mobile overview watching badge', () => { expect(model.needsYou[0].watching).toBe('2 shells'); }); - it('says one word and puts the detail in the tooltip', () => { + it('says one word and puts the detail where every surface can reach it', () => { const app = loadOverviewApp(); const badge = app._buildWatchingBadge('1 monitor', 'mobile-overview-pill'); expect(badge.className).toBe('mobile-overview-pill mobile-overview-pill--watching'); expect(badge.textContent).toBe('watching'); expect(badge.title).toBe('Still running in the background: 1 monitor'); + // A phone has no hover target and a screen reader reads neither the class nor the + // tooltip, so the label has to be here too or this surface says only "watching". + expect(badge.attrs['aria-label']).toBe('Still running in the background: 1 monitor'); }); it('takes the pill class of whichever surface asks for it', () => { diff --git a/test/session-sidebar-ux.test.ts b/test/session-sidebar-ux.test.ts index f74c1ace..6b2ed047 100644 --- a/test/session-sidebar-ux.test.ts +++ b/test/session-sidebar-ux.test.ts @@ -97,6 +97,24 @@ describe('watching badge on a rich session row', () => { ); }); + it('escapes the label everywhere it reaches markup', () => { + // `watching` is pane-derived and a config-supplied pattern decides what its capture + // group holds, so every interpolation of it into HTML has to go through escapeHtml(). + // The row is installed with innerHTML, which makes an unescaped quote in that + // attribute an injection rather than a cosmetic bug. + expect(app).toContain('${richRow.createdAt}:${escapeHtml(richRow.watching)}"'); + expect(app).not.toContain('${richRow.createdAt}:${richRow.watching}"'); + }); + + it('words the tooltip exactly as the phone overview does', () => { + // Both files build this sentence themselves, deliberately, so that a stale cached + // module still renders a complete row. Substring-matching the prefix would let the + // two drift; the whole sentence is what has to agree. + const overview = readFileSync(resolve(publicDir, 'mobile-overview.js'), 'utf8'); + expect(overview).toContain("'Still running in the background: ' + label"); + expect(app).toContain('`Still running in the background: ${row.watching}`'); + }); + it('colours it with the accent, never with the two colours that mean a human is needed', () => { const rule = styles.slice(styles.indexOf('.tab-pill--watching')); const block = rule.slice(0, rule.indexOf('}')); diff --git a/test/session-watching.test.ts b/test/session-watching.test.ts index e539cf94..fc8f4d03 100644 --- a/test/session-watching.test.ts +++ b/test/session-watching.test.ts @@ -167,6 +167,26 @@ describe('watchingLabel', () => { expect(watchingLabel(`${chip}\n${below}\n`, CLAUDE_WATCHING)).toBeNull(); }); + it('refuses a chip on the row above the footer, which the agent can write', () => { + // The status line is one row up, its text comes from a `statusLine` command, and a + // session running with permissions bypassed can write that command into + // `.claude/settings.json` in its own workspace. The window is what keeps that row + // out, so this is the test that would fail if somebody widened it. + const forged = pane('⏵⏵ bypass permissions on (shift+tab to cycle) · ← for agents').replace( + ' ~/innovi/gtd-board [main] Opus 5 ctx: 11%', + ' ~/innovi/gtd-board [main] Opus 5 ctx: 11% · 1 monitor' + ); + expect(watchingLabel(forged, CLAUDE_WATCHING)).toBeNull(); + // And with the window widened by one, the same screen does match — which is the + // whole reason the default is one row. + expect(watchingLabel(forged, CLAUDE_WATCHING, 2)).toBe('1 monitor'); + }); + + it('keeps Claude on the default window, because its chip is the last row', () => { + expect(getCli('claude')?.capabilities.workDetect?.watchingLines).toBeUndefined(); + expect(WATCHING_TAIL_LINES).toBe(1); + }); + it('refuses a label the footer did not separate, which is the injection guard', () => { // The pattern anchors on the `·` the footer joins its items with. Without that // anchor an agent could silence its own idle alert by printing the words, since the @@ -296,14 +316,34 @@ describe('the row Codex draws', () => { expect(CODEX_TAIL).toBeGreaterThanOrEqual(3); }); - it('refuses the same words in the transcript, which is the injection guard', () => { - // ` · /ps to view` is chrome: only the CLI offers that slash command. Without the - // anchor an agent could print the sentence and silence itself. - const claim = CODEX_STOPPED.replace( + it('refuses a mention that is not the whole row', () => { + // The pattern matches Codex's row end to end, so prose about background terminals — + // including prose quoting part of the row — is not enough. + for (const line of [ + '• I left 1 background terminal running for you.', + ' 1 background terminal running · /ps to view', + ' see: 1 background terminal running · /ps to view · /stop to close', + ]) { + const claim = CODEX_STOPPED.replace('• Stopping all background terminals.', line); + expect(watchingLabel(claim, CODEX_WATCHING, CODEX_TAIL)).toBeNull(); + } + }); + + it('CAN be forged by Codex own output, and is contained by Codex having no hooks', () => { + // Codex's row is third from the bottom only while a terminal runs; with none running + // that slot is the last row of the transcript, which the agent writes. Matching the + // complete row raises the bar but closes nothing, so this test states the limitation + // rather than a protection the code does not have. + const forged = CODEX_STOPPED.replace( '• Stopping all background terminals.', - '• I left 1 background terminal running for you.' + ' 1 background terminal running · /ps to view · /stop to close' ); - expect(watchingLabel(claim, CODEX_WATCHING, CODEX_TAIL)).toBeNull(); + expect(watchingLabel(forged, CODEX_WATCHING, CODEX_TAIL)).toBe('1 background terminal'); + + // What makes that cost a wrong badge and nothing more: no hook event from a codex + // session reaches the approvals inbox, so there is no idle item to pre-acknowledge + // and no alert to silence. A CLI that gains hook signals needs a harder anchor first. + expect(getCli('codex')?.capabilities.hooks).toBe('none'); }); }); diff --git a/test/tui/tui-approvals.test.ts b/test/tui/tui-approvals.test.ts index f9072f80..81d0585c 100644 --- a/test/tui/tui-approvals.test.ts +++ b/test/tui/tui-approvals.test.ts @@ -59,6 +59,24 @@ describe('approvalCard', () => { expect(card.hint).toBe('p to reply'); }); + it('says why an acknowledged idle prompt is quiet, in the drawer own words', () => { + // The session is watching work it started itself. Asking for a reply would be the + // same false alarm the acknowledgement exists to remove, so the card states the + // reason and drops out of the warning vocabulary — while staying answerable. + const card = approvalCard( + item({ + kind: 'idle', + message: 'Claude is waiting for your input', + options: undefined, + acknowledgedAt: 1_700_000_000_000, + acknowledgedReason: 'watching 1 monitor', + }) + ); + expect(card.tone).toBe('info'); + expect(card.title).toBe('quiet, watching 1 monitor'); + expect(card.hint).toBe('p to reply'); + }); + it('drops the approve/deny-only hint when the frame did not parse', () => { const card = approvalCard(item({ options: undefined })); expect(card.options).toEqual([]); @@ -81,6 +99,12 @@ describe('approvalTone', () => { expect(approvalTone(item({ kind: 'question' }))).toBe('err'); expect(approvalTone(item({ kind: 'idle' }))).toBe('warn'); }); + + it('is neither for a prompt that opened acknowledged', () => { + expect(approvalTone(item({ kind: 'idle', acknowledgedReason: 'watching 2 shells' }))).toBe('info'); + // A dialog stays red whatever else the session started. + expect(approvalTone(item({ kind: 'permission', acknowledgedReason: 'watching 2 shells' }))).toBe('err'); + }); }); describe('approvalAnswerForKey', () => { diff --git a/test/tui/tui-render.test.ts b/test/tui/tui-render.test.ts index 43e5f2c9..fcb3df27 100644 --- a/test/tui/tui-render.test.ts +++ b/test/tui/tui-render.test.ts @@ -11,6 +11,7 @@ import { charWidth, stripStyles, toDisplayLines, visibleWidth } from '../../src/ import { composerMove, createComposer } from '../../src/tui/tui-composer.js'; import { computeLayout, needsBanner } from '../../src/tui/tui-layout.js'; import { createTuiModel, type TuiModelStore } from '../../src/tui/tui-model.js'; +import type { ApprovalItem } from '../../src/web/approval-inbox.js'; import { composerCursorCell, detectGlyphTier, @@ -19,6 +20,7 @@ import { formatPlanUsage, formatTokens, glyphsFor, + pendingApprovalCount, renderFrame, rowLabel, type TuiRenderOptions, @@ -682,3 +684,33 @@ describe('the unicode glyph set is safe to render', () => { expect(every.join('')).not.toContain('\u270B'); }); }); + +describe('pendingApprovalCount', () => { + const prompt = (over: Partial): ApprovalItem => ({ + id: 'bbb2:1', + sessionId: 'bbb2', + sessionName: 'w6-docs', + kind: 'idle', + createdAt: NOW - 30_000, + ...over, + }); + + it('counts a prompt nobody has seen', () => { + const model = fixture(); + model.setApprovals([prompt({})]); + expect(pendingApprovalCount(model)).toBe(1); + }); + + it('does not count one whose alert is already spent', () => { + // Both ways an item gets acknowledged: a human opening the session elsewhere, and the + // inbox opening it that way for a session watching its own background work. The row + // has left NEEDS YOU by the same flag, so a number in the header would point at a + // group the reader can see is empty. + const model = fixture(); + model.setApprovals([prompt({ acknowledgedAt: NOW - 20_000 })]); + expect(pendingApprovalCount(model)).toBe(0); + + model.setApprovals([prompt({ acknowledgedAt: NOW - 20_000, acknowledgedReason: 'watching 1 monitor' })]); + expect(pendingApprovalCount(model)).toBe(0); + }); +});