From 161f1da2eb109e8a4abfc0d707bb6259e856210c Mon Sep 17 00:00:00 2001 From: Codeman maintainer Date: Sun, 9 Aug 2026 18:03:13 +0200 Subject: [PATCH] feat: Read My Mind phase 1, per-case intent profiles (capture + API + skill) Per-case profiles of user intent (docs/readmymind-plan.md): user-stated goals plus the user's recently submitted prompts, captured from the Claude session transcript behind the new synced readMyMindEnabled setting (default OFF). - intent-store.ts: keyed by owner + realpath(workingDir), FIFO/size caps, consecutive-dupe collapse, atomic 0600 writes to ~/.codeman/intents.json - transcript-watcher.ts: new transcript:user_prompt event for typed user turns (tool_result-only entries stay silent); capture wiring in server.ts is claude-only and gated on the setting per event - readmymind-routes.ts: GET/PUT/DELETE /api/sessions/:id/intent, ownership via findSessionOrFail, strict Zod schema - agent skill: SKILL.md recipe + endpoints.md rows so agents can read and record intent (PUT replaces: read + merge; never delete unprompted) - groundwork for the phase-2 predictor button; nothing is ever auto-sent Co-Authored-By: Claude Fable 5 --- .changeset/1d154a73.md | 5 + CLAUDE.md | 8 +- docs/api-reference.md | 24 +++ docs/readmymind-plan.md | 140 ++++++++++++++++ skills/codeman/SKILL.md | 18 ++ skills/codeman/reference/endpoints.md | 3 + src/intent-store.ts | 233 ++++++++++++++++++++++++++ src/transcript-watcher.ts | 12 ++ src/types/index.ts | 1 + src/types/intent.ts | 31 ++++ src/web/routes/index.ts | 1 + src/web/routes/readmymind-routes.ts | 50 ++++++ src/web/schemas.ts | 17 ++ src/web/server.ts | 27 ++- test/intent-store.test.ts | 187 +++++++++++++++++++++ test/routes/readmymind-routes.test.ts | 148 ++++++++++++++++ test/transcript-watcher.test.ts | 82 +++++++++ 17 files changed, 983 insertions(+), 4 deletions(-) create mode 100644 .changeset/1d154a73.md create mode 100644 docs/readmymind-plan.md create mode 100644 src/intent-store.ts create mode 100644 src/types/intent.ts create mode 100644 src/web/routes/readmymind-routes.ts create mode 100644 test/intent-store.test.ts create mode 100644 test/routes/readmymind-routes.test.ts diff --git a/.changeset/1d154a73.md b/.changeset/1d154a73.md new file mode 100644 index 00000000..e77ac314 --- /dev/null +++ b/.changeset/1d154a73.md @@ -0,0 +1,5 @@ +--- +"aicodeman": minor +--- + +Read My Mind phase 1: per-case intent profiles (docs/readmymind-plan.md). Codeman can now capture the prompts a user actually submits (from the Claude session transcript, opt-in via the new synced readMyMindEnabled setting, default OFF) into a per-case intent profile alongside user-stated goals, stored in ~/.codeman/intents.json (mode 0600, never searched). New endpoints GET/PUT/DELETE /api/sessions/:id/intent (ownership-scoped, strict schemas), a transcript:user_prompt event on TranscriptWatcher, and agent-skill coverage (SKILL.md recipe + endpoints.md rows) so agents can read and record the user's intent. Groundwork for the phase-2 predictor button: nothing is ever auto-sent. diff --git a/CLAUDE.md b/CLAUDE.md index 5af91106..094efaa4 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -153,7 +153,7 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph | **Agents** | `src/subagent-watcher.ts` ★, `team-watcher`, `bash-tool-parser`, `transcript-watcher`, `workflow-run-watcher` | `workflow-run-watcher` is STANDALONE and never touches `subagent-watcher` | | **AI** | `src/ai-checker-base.ts`, `ai-idle-checker.ts`, `ai-plan-checker.ts` | | | **Tasks** | `src/task.ts`, `task-queue.ts`, `task-tracker.ts` | | -| **State** | `src/state-store.ts`, `run-summary.ts`, `session-lifecycle-log.ts` | | +| **State** | `src/state-store.ts`, `run-summary.ts`, `session-lifecycle-log.ts`, `intent-store.ts` | | | **Infra** | `src/hooks-config.ts`, `push-store`, `tunnel-manager`, `image-watcher`, `file-stream-manager`, `remote-hosts` + `remote-reconnect` (pure), `docker-hosts` + `docker-export` | Remote/docker case overlays; see Key Patterns | | **Web tabs** | `src/webview-store.ts`, `webview-capabilities.ts`, `src/web/webview-proxy.ts` (pure), `src/web/routes/webview-routes.ts` | Dashboard URLs as tabs; NOT a SessionMode | | **Search** | `src/search-service.ts` | Pure in-memory core for `GET /api/search` | @@ -210,6 +210,8 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph **Approvals Inbox** (cross-session queue of prompts waiting on a human; `approvalsInboxEnabled`, SYNCED, default OFF: every surface is opt-in; only the store and answer endpoints run regardless, so flipping it ON shows anything already pending): `web/approval-inbox.ts` is a `sessionWaits`-style singleton fed by `/api/hook-event`, holding at most ONE item per session (a new prompt supersedes), claude-mode only, in-memory. Cards are answered via `POST /api/approvals/:id/answer`, which sends a digit / Esc / idle-prompt text through `writeViaMux` (menu answers never carry `\r`). ⚠️ `option` digits are accepted ONLY when they match options parsed from the captured pane frame, and the answer path RE-CAPTURES the pane first (a dialog that no longer parses on screen means the keystroke would land in the composer, so refuse with 409). ⚠️ Resolution on the heuristic `working` signal is restricted to `idle` items; permission/question items clear only on definitive signals (`stop`, `elicitation_complete`/`elicitation_response`, exit/delete, answer, supersede, 12h TTL). The frontend seeds from `GET /api/approvals` in `handleInit` (which is what makes tab alerts survive reloads), but only with the setting ON; push Approve/Deny buttons are also gated on it (`sendPushNotifications` strips `actions`/`approvalId` when OFF) and are answered from `sw.js` directly so they work with no tab open. Surfaces (all gated on the setting): header bell (marker-hidden until count > 0, phones never show it) + drawer (`approvals-ui.js`), phone overview NEEDS YOU answer strips (`mobile-overview.js`). Design: `docs/approvals-inbox-plan.md`. +**Read My Mind intent profiles** (phase 1 of `docs/readmymind-plan.md`; `readMyMindEnabled`, SYNCED, default OFF): per-CASE profiles (user-stated `goals` + the user's recent real prompts), keyed by owner + realpath(workingDir) so they survive `/clear`/respawns and multi-user scoping is structural. Capture rides the transcript (`transcript:user_prompt` from `transcript-watcher.ts`), NOT the input paths: `POST /input` sees only programmatic prompts and the WS channel is raw keystrokes. The listener lives inside `startTranscriptWatcher()`'s `if (!watcher)` block (outside it would duplicate per hook event) and is claude-only + gated on the setting per event. Store: `src/intent-store.ts` singleton, `intents.json` written 0600 tmp+rename (prompts can contain secrets; never fed to `/api/search`). Endpoints: GET/PUT/DELETE `/api/sessions/:id/intent` (`readmymind-routes.ts`, ownership via `findSessionOrFail` WITH `req`). The predictor/button are phase 2; nothing auto-sends, ever. + **Agent Teams**: `TeamWatcher` polls `~/.claude/teams/`, matches to sessions via `leadSessionId`. Teammates are in-process threads appearing as subagents. Enable: `CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1`. See `docs/agent-teams/`. **Circuit breakers**: the Ralph breaker prevents respawn thrashing (`CLOSED` → `HALF_OPEN` → `OPEN`; reset via `/api/sessions/:id/ralph-circuit-breaker/reset`). **Distinct: the PTY-exit breaker** (`session-pty-exit-breaker.ts`) trips after repeated rapid PTY exits and blocks auto-restarts. ⚠️ It resets ONLY via an explicit `{clearBreaker:true}` body on `POST /api/sessions/:id/interactive`; the frontend's auto-reattach in `selectSession()` sends no body and must never clear it. → [architecture-invariants#circuit-breakers-ralph--pty-exit](docs/architecture-invariants.md#circuit-breakers-ralph-and-pty-exit) @@ -304,7 +306,7 @@ Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. L ### API Routes -~200 handlers across 22 route files in `src/web/routes/`: system (45), sessions (34), cases (27), files (16), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), approvals (3), me (2), teams (2), search (1), hooks (1), clipboard (1), status-telemetry (1), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details. +~200 handlers across 23 route files in `src/web/routes/`: system (45), sessions (34), cases (27), files (16), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), approvals (3), readmymind (3), me (2), teams (2), search (1), hooks (1), clipboard (1), status-telemetry (1), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details. **HTTP contract** (stable since 0.9.x, see `docs/versioning-policy.md`; full envelope/status/error-code/SSE spec in `docs/api-reference.md`): responses use the `ApiResponse` envelope — `{ success: true, data? }` or `{ success: false, error, errorCode }` (`src/types/api.ts`). `/api/v1/*` is a versioned alias of `/api/*` (URL rewrite in `server.ts`). @@ -322,7 +324,7 @@ Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. L ## State Files -All in `~/.codeman/`: `state.json` (sessions, settings, respawn, orchestrator, cron jobs/runs), `mux-sessions.json` (tmux recovery), `settings.json` (user prefs), `push-keys.json` + `push-subscriptions.json`, `session-lifecycle.jsonl` (audit log), `update-status.json` (self-updater progress, polled across the service restart), `linked-cases.json`, `webviews.json` (saved web-tab dashboard URLs), `remote-hosts.json` + `remote-cases.json`, `docker-hosts.json` + `docker-cases.json` + `docker-exports/`, `subagent-window-states.json` + `subagent-parents.json` (subagent window layout, GET/PUT `/api/subagent-window-states`/`-parents`), `hook-secret` (per-instance), `users.json` (multi-user, mode 0600) + `admin-audit.jsonl`, `certs/` (self-signed TLS for `--https`), `.env` (CODEMAN_USERNAME/PASSWORD fallback for the `codeman attach` CLI). Transient: `self-update-runner.sh`. Multi-user case spaces live OUTSIDE the data dir at `~/codeman-users//cases` (shared across instances like `~/codeman-cases`, override `CODEMAN_USER_SPACES_DIR`). +All in `~/.codeman/`: `state.json` (sessions, settings, respawn, orchestrator, cron jobs/runs), `mux-sessions.json` (tmux recovery), `settings.json` (user prefs), `push-keys.json` + `push-subscriptions.json`, `session-lifecycle.jsonl` (audit log), `update-status.json` (self-updater progress, polled across the service restart), `linked-cases.json`, `webviews.json` (saved web-tab dashboard URLs), `remote-hosts.json` + `remote-cases.json`, `docker-hosts.json` + `docker-cases.json` + `docker-exports/`, `subagent-window-states.json` + `subagent-parents.json` (subagent window layout, GET/PUT `/api/subagent-window-states`/`-parents`), `hook-secret` (per-instance), `users.json` (multi-user, mode 0600) + `admin-audit.jsonl`, `intents.json` (Read My Mind intent profiles, mode 0600), `certs/` (self-signed TLS for `--https`), `.env` (CODEMAN_USERNAME/PASSWORD fallback for the `codeman attach` CLI). Transient: `self-update-runner.sh`. Multi-user case spaces live OUTSIDE the data dir at `~/codeman-users//cases` (shared across instances like `~/codeman-cases`, override `CODEMAN_USER_SPACES_DIR`). **Generated top-level dirs** (all gitignored — don't edit or commit): `dist/` (esbuild output), `out/`, `coverage/`, `test-results/`, `tmp/`, `screenshots-echo-diag/`. The committed gesture bundle (`src/web/public/gesture/gesture-codeman.js`) IS tracked, but its runtime wasm/model assets (`src/web/public/gesture/wasm/`, `*.task`) are fetched and gitignored. diff --git a/docs/api-reference.md b/docs/api-reference.md index 9244985c..f22a9fad 100644 --- a/docs/api-reference.md +++ b/docs/api-reference.md @@ -434,6 +434,30 @@ re-captured), `approval:resolved` (`{ id, sessionId, kind, resolution }` with `resolution` one of `answered | resolved_in_terminal | superseded | session_ended | dismissed | expired`). +## Read My Mind intent profiles + +Per-case profiles of what the user is trying to accomplish: user/agent-stated +goals plus the user's recently submitted prompts, captured from the Claude +session transcript while the opt-in `readMyMindEnabled` setting is on (default +OFF). Keyed by owner + workingDir, so the profile survives `/clear`, respawns, +and session churn. Stored in `~/.codeman/intents.json` (mode 0600); never fed +into `/api/v1/search`. Design: [`readmymind-plan.md`](readmymind-plan.md). + +- `GET /api/v1/sessions/:id/intent` -> `{ intent: IntentProfile }` for the + session's case. `IntentProfile`: `{ key, workingDir, updatedAt, goals, + recentPrompts: { ts, sessionId, text }[] }` (prompts oldest first, FIFO cap + 50, each <= 500 chars). A case with nothing recorded answers an empty + profile with `updatedAt: 0`; nothing is persisted by reads. +- `PUT /api/v1/sessions/:id/intent` with `{ goals }` (<= 8192 chars, strict + schema) replaces the goals text and answers the updated profile. + `400 INVALID_INPUT` on over-long or unknown fields. +- `DELETE /api/v1/sessions/:id/intent` -> `{ deleted: boolean }` forgets the + case's profile entirely. + +All three enforce session ownership in multi-user mode; a foreign session id +answers `404 NOT_FOUND` (no existence leak), and profiles of two owners of the +same directory are distinct by construction. + ## Authentication Optional HTTP Basic (`CODEMAN_USERNAME`/`CODEMAN_PASSWORD`) → opaque diff --git a/docs/readmymind-plan.md b/docs/readmymind-plan.md new file mode 100644 index 00000000..d263bf07 --- /dev/null +++ b/docs/readmymind-plan.md @@ -0,0 +1,140 @@ +# Read My Mind (design) + +A 🧠 button that predicts the prompt you were about to type. Codeman keeps a per-case **intent profile** (your stated goals plus the real prompts you recently sent), feeds it and the live pane tail to a one-shot `claude -p`, and shows the predicted next prompt in a plan-mode-style approval dialog: **Send** / **Rethink** (with an optional steer note) / **Insert** (drop it on the composer to edit) / **Dismiss**. It is also a skill surface: the agent can read the intent profile, record intentions, and request a prediction over the HTTP API. Suggestions are **never auto-sent**; the human click is the boundary. + +## UX flow + +1. User hits 🧠 (desktop header button; phone: keyboard-accessory key). +2. Modal opens with a spinner, then the top suggestion in an editable single-line field, rationale below it, up to 2 alternates as tappable rows. +3. Buttons: **Send** (submits with `\r`), **Insert** (sends without `\r`, so the text sits unsubmitted on the CLI composer for editing, a documented mechanism), **Rethink** (optional free-text steer, e.g. "no, I meant the mobile bug", re-runs with the rejected suggestions included), **Dismiss**. +4. Accepted prompts flow back into the intent history like any other sent prompt, so the profile self-corrects. + +## Scope (v1) + +- Claude mode only (capture rides Claude transcripts; external CLIs have no transcript watcher). Mirrors the approvals-inbox scoping. +- Opt-in: `readMyMindEnabled`, synced, default **OFF**. While OFF: no capture, no UI surfaces. Privacy first, and every press costs real tokens. +- One prediction in flight per session; the button disables while checking. +- Sync request/response (the predictor takes 5-30s; agent-wait long-polls already hold requests longer). No new SSE events in v1. + +## Data model + +Per case, not per session: intentions outlive `/clear` and respawns. + +```ts +interface IntentProfile { + key: string; // sha256(owner + ':' + realpath(workingDir)).slice(0, 16) + workingDir: string; + updatedAt: number; + goals: string; // freeform markdown, user/agent editable, ≤ 8 KB + recentPrompts: { ts: number; sessionId: string; text: string }[]; // FIFO cap 50, each ≤ 500 chars +} +``` + +Storage: `dataPath('intents.json')`, written mode 0600 (prompts can contain secrets; same posture as `users.json`). Never enters the `/api/search` index. Add to the CLAUDE.md State Files list. + +## Intent capture + +**Source: the session transcript, not the input paths.** `POST /api/sessions/:id/input` sees only programmatic input, and the WS channel delivers raw keystrokes (`session.write(msg.d)`), so neither yields clean submitted prompts. Claude's own JSONL transcript records every user turn as structured text, and `transcript-watcher.ts` already tails it. Add a `userPrompt` event there: + +- Emit for `type: 'user'` entries whose content is a string or contains a text block; skip entries that are only `tool_result` blocks (tool results are wrapped as user messages). +- Skip `` / `` tagged entries (local slash-command echo, not intent). +- Skip texts < 3 chars (menu digits, Esc artifacts), truncate to 500, drop consecutive duplicates ("continue" spam from auto-resume stays but dedupes). + +`IntentStore` (new `src/intent-store.ts`, pure core + IO wrapper, in the style of `session-order.ts`) subscribes via session wiring, gated on the setting resolved from **merged** settings per the partial-PUT rule. + +## Context assembly (how the mind reading actually works) + +The quality of the suggestion is decided before the model ever runs, by what we put in front of it. A new pure function `buildPredictionContext()` (in `src/readmymind-context.ts`, unit-testable with fixtures, no IO of its own; collectors inject their data) assembles a budgeted, priority-ordered prompt from every signal Codeman already has: + +| # | Source | What it contributes | Cap | +| - | ------ | ------------------- | --- | +| 1 | **Pending dialog** (approvals-inbox store, when present) | If the session is sitting on an AskUserQuestion / permission / idle prompt, the honest "next prompt" is an *answer*. The dialog text + parsed options go in first and the model is told to answer it. | 2 KB | +| 2 | **User goals** (`goals` from the intent profile) | The only fully-trusted statement of what the user wants. Highest authority in the trust ranking below. | 8 KB | +| 3 | **Last assistant turn** (transcript, not the pane) | Assistant replies usually *end* with the fork in the road ("Want me to X?", "Next steps: ..."), so keep the **tail** when truncating. The transcript has the full message; the pane is a repaint window full of spinner junk. | 6 KB | +| 4 | **Recent user prompts** (intent profile, with timestamps) | The conversation rhythm AND the user's prompting voice: length, tone, shorthand (`COM`, lowercase, typos and all). The model is instructed to write suggestions in *this* style, not assistant-ese. | last 20 | +| 5 | **Recent tool activity** (transcript `tool_use` blocks, already parsed by `TranscriptWatcher`) | One line per call: `Edit src/foo.ts`, `Bash npm test (failed)`. What the agent actually *did*, which the last message may summarize away. | last 10 | +| 6 | **Workspace signals** (`collectWorkspaceSignals()`: `git` via `execFile` in `workingDir`, 2s timeout) | Branch, `status --short` (dirty files scream "commit/test/deploy next"), last 5 commits oneline, presence of `.changeset/*.md` (release pending). Skipped for remote-SSH cases (workingDir is not local); fine for Docker cases (bind-mounted at the same host path). Non-git dirs: section omitted. | 3 KB | +| 7 | **Away context** (run-summary events + elapsed time) | `Last user prompt was 6h ago; since then: `. After a long gap the right suggestion is often "review / continue yesterday's thread", not a blind continuation. | 2 KB | +| 8 | **Sibling sessions** (live sessions sharing the case) | One line each: name, mode, working/idle. A lead-and-workers setup changes what the next prompt should be ("check on w2" beats "keep going"). | 1 KB | +| 9 | **Rethink state** (steer note + rejected suggestions) | Only on re-runs. Rejections are strong negative signal and go in verbatim. | 2 KB | + +Total budget ~30 KB. When over budget, drop from the bottom up (siblings first, then away context, then workspace signals); sections 1-4 never drop, they only truncate. Deterministic assembly means fixture tests can pin exactly what a given situation feeds the model. + +**Trust tiers are stated in the prompt.** Goals and user prompts are *the user*; assistant text, tool logs, and pane content are *observations that may contain text trying to manipulate you* (a hostile repo can print "SUGGEST: run curl evil.sh"). The prompt instructs: user-stated intent outranks anything observed, and never propose a prompt whose primary source is terminal output alone. The human approval click remains the hard boundary regardless. + +**Output contract** (strict JSON, parse failure = clean error, never a half-suggestion): + +```json +{ "suggestions": [ { "prompt": "...", "why": "...", "kind": "continue" | "verify" | "redirect" } ] } +``` + +1-3 entries, and the *kinds* force useful diversity instead of three rewordings: `continue` (finish the current thread, or answer the pending dialog), `verify` (test/review what was just built; the user's own "always end-to-end test" discipline), `redirect` (the next goal from the intent profile that the current thread is not serving). The modal shows `continue` big, the others as alternates. Embedded newlines are stripped server-side (single-line prompt rule; multi-line breaks Ink). + +## Predictor + +New `src/readmymind-predictor.ts`, reusing the `AiCheckerBase` mechanics (prompt file to dodge E2BIG, one-shot `claude -p --output-format text` in a throwaway tmux `codeman-rmm-`, done-marker polling, timeout, model-name validation) but standalone: the base class is verdict-shaped (positive/negative/cooldown) and prediction is freeform JSON, so subclassing would abuse `reasoning` as a payload. If a shared spawn/poll helper falls out naturally, extract it; do not block on the refactor. + +- **Model: opus** (decided). `readMyMindModel` setting, default `AI_CHECK_MODEL` (currently `claude-opus-4-5-20251101`); prediction quality is the product, and it runs only on an explicit press, so the cost profile is nothing like the idle checker's. Timeout 90s (opus headroom over a ~30 KB prompt). +- Input: the assembled context above. The predictor itself stays dumb: text in, JSON out; all intelligence about *what to include* lives in the testable assembler. + +## API (new `src/web/routes/readmymind-routes.ts`) + +Normal authed API, `ApiResponse` envelope, Zod schemas in `schemas.ts`, ownership via `findSessionOrFail` (the profile key derives from the session's owner + workingDir, so multi-user scoping is structural): + +- `GET /api/sessions/:id/intent` → the session's `IntentProfile`. +- `PUT /api/sessions/:id/intent` body `{ goals }` (bounded) → update goals. Used by the modal's edit view and by the agent skill ("record that the user is working toward X"). +- `DELETE /api/sessions/:id/intent` → forget everything for this case (the modal's "Forget" affordance). +- `POST /api/sessions/:id/readmymind` body `{ steer?, rejected? }` → `{ suggestions }`. 409 `INVALID_STATE` while a prediction is already running for the session; claude-mode sessions only (400 otherwise, mirroring wait-signal gating). + +## Frontend + +New module `readmymind-ui.js` (@loadorder 11.3, after panels-ui.js), prettier-formatted. + +- **Desktop**: header button `btn-readmymind`, default-hidden via marker class `btn-readmymind--hidden` (the `!important` display rules require the marker-class pattern), shown by `applyHeaderVisibilitySettings()` when the setting is ON. Off phones per `test/mobile-header-buttons-policy.test.ts`. +- **Phone**: a 🧠 key on the keyboard accessory bar (that bar is where input helpers live, and phones are where typing hurts most). Opens the same modal. Modal z-index respects the ≤768px layer rules (1300+). +- **Send** goes server-side: `POST /api/sessions/:id/input` with `\r` appended. Deliberately NOT the browser keystroke path, so the `sendEnterKey` / local-echo-overlay trap never applies (the modal is UI chrome, not terminal typing). **Insert** is the same POST without `\r`. +- i18n strings registered (en + zh-CN); suggestion text itself carries `data-i18n-skip`. + +## Skill integration + +The user-facing promise: the button is also a skill. Extend `skills/codeman`: + +- New section "Read My Mind: intent + prediction" with the three intent verbs (read profile, append/replace goals, predict) and the guard notes (single-line prompts, never auto-send to another session without the user asking). +- Update `reference/endpoints.md` (the endpoints.md drift test pins this). +- The auto-injected case copy heals via the existing marker-owned `applyAgentSkill` mechanism; nothing new needed there. + +Agent use cases this unlocks: a lead session records intentions as the user states them ("remember: shipping 1.16 is the goal"), and a returning user gets a prediction grounded in what the agent knew, not just raw prompt history. + +## Security / privacy + +- **The human gate is the injection mitigation**: pane output (attacker-influenceable) flows into the predictor, so its output is only ever *proposed*, rendered as text (`textContent`), and sent solely by an explicit user click. No auto-send path exists, including for the skill. +- Intent data: 0600 file, bounded fields, per-owner keys, endpoints ownership-checked, excluded from search, cleared via DELETE. +- Predictor spawns with the user's own credentials exactly like the AI idle/plan checkers; model name shell-validated the same way. +- Setting OFF stops capture immediately; existing data stays until DELETE (explicit, not silent). + +## Tests + +- `test/intent-store.test.ts`: key derivation, caps/FIFO, consecutive-dupe skip, tag/tool_result filtering fixtures, 0600 mode, multi-user key separation. +- `test/readmymind-context.test.ts`: fixture scenarios pinning the assembled prompt: pending-dialog-first ordering, tail-keeping truncation of the assistant turn, budget drop order (siblings before workspace signals), remote-case git skip, trust-tier framing present, rejected suggestions included only on rethink. +- `test/readmymind-predictor.test.ts`: strict JSON parse, garbage output → error result, newline stripping, `kind` validation, rejected-suggestions threading into the prompt. +- `test/routes/readmymind-routes.test.ts` (`app.inject`): CRUD round-trip, predict with a stubbed predictor, 409 while in flight, non-claude 400, ownership 404, Send/Insert byte assertions via the test-PTY echo (`\r` present vs absent). +- Transcript capture: extend the transcript-watcher fixtures with user-turn entries. + +## Phases + +1. **Intent store + capture + intent endpoints + skill docs.** Immediately useful to agents even before any UI exists. +2. **Context assembler + predictor + predict endpoint + desktop button/modal.** The feature as pitched. The assembler ships with all collectors it can serve from day one (transcript, intent, git, run-summary, siblings); the approvals collector activates when PR #245 lands. +3. **Phone accessory key, rethink steering, alternates row.** +4. Explicitly later: proactive predict-on-idle (ghost suggestion chip), auto-compaction of `recentPrompts` into `goals` via a cheap model, codex/gemini capture, cross-case "global" intent. + +## Open questions + +- Should Rethink's rejected-suggestion memory persist across modal closes, or reset each open? +- Is a composer-adjacent placement (next to the toolbar Run controls) better than the header for discoverability? +- Pending-dialog input (source #1) consumes the approvals-inbox store (PR #245, merged): the phase-2 collector reads pending items directly from `src/approval-inbox.ts`. + +## Docs + +- CLAUDE.md: Key Patterns entry, State Files (`intents.json`), frontend load order, route count. +- `docs/api-reference.md`: four endpoints (additive under the 0.9.x contract). +- `skills/codeman/reference/endpoints.md`: new rows (drift-test enforced). diff --git a/skills/codeman/SKILL.md b/skills/codeman/SKILL.md index c5496d7f..abcd77e4 100644 --- a/skills/codeman/SKILL.md +++ b/skills/codeman/SKILL.md @@ -392,6 +392,24 @@ is parked resolves it within ~3 s. A session deleted mid-wait resolves in ~1 s. delete_session "$SID" ``` +**Read My Mind: read and record the user's intent.** Each case has an intent +profile: user-stated goals plus the user's recent real prompts (captured +server-side while the opt-in `readMyMindEnabled` setting is on). Read it to +ground your work in what the user actually wants; write it when the user states +an intention worth remembering ("the goal is shipping 1.17"): + +```bash +"${CURL[@]}" "$API/api/v1/sessions/$SELF/intent" | jq '.data.intent' +"${CURL[@]}" -X PUT -H 'Content-Type: application/json' \ + -d '{"goals":"shipping 1.17; mobile polish next"}' "$API/api/v1/sessions/$SELF/intent" +``` + +⚠️ PUT **replaces** the whole goals text: read it first and merge, never +blind-write. Never write goals the user did not state, and never delete the +profile (`DELETE .../intent`) unless the user asks: it is their memory, not +yours. Older servers 404 these routes; treat that as "feature absent", not an +error. + Everything else (endpoint tables, per-mode signal table, error codes, capacity limits, Docker/remote caveats): [reference/endpoints.md](reference/endpoints.md). Fan-out orchestration and blocked-worker handling: diff --git a/skills/codeman/reference/endpoints.md b/skills/codeman/reference/endpoints.md index 8d988d98..29217a33 100644 --- a/skills/codeman/reference/endpoints.md +++ b/skills/codeman/reference/endpoints.md @@ -47,6 +47,9 @@ read the status with `-w '%{http_code}'` and the raw body before assuming a bug. | full tmux scrollback (context bomb; post-mortems only) | `GET /api/v1/sessions/:id/terminal?full=1` | | background agents, one session | `GET /api/v1/sessions/:id/subagents` | | background agents, global list | `GET /api/v1/subagents` (admin-only in multi-user mode) | +| the case's intent profile (Read My Mind: user goals + recent real prompts) | `GET /api/v1/sessions/:id/intent` → `.data.intent.{goals,recentPrompts}` (empty with `updatedAt: 0` until something is recorded) | +| replace the user-goals text on the case's intent profile | `PUT /api/v1/sessions/:id/intent` body `{"goals":"…"}` (≤ 8192 chars, strict schema; REPLACES the text, read + merge first) | +| forget the case's intent profile (only when the user asks) | `DELETE /api/v1/sessions/:id/intent` → `.data.deleted` | | server status / version | `GET /api/v1/status` → `.data.version` | | delete one session (yours only, via `delete_session`) | `DELETE /api/v1/sessions/:id` — never call it bare; the fail-closed helper in SKILL.md §0 is the only self-protection that exists. Answers `{"success":true,"data":{}}`: an **empty** body is the success signal, there is nothing to read back | diff --git a/src/intent-store.ts b/src/intent-store.ts new file mode 100644 index 00000000..d1592bf9 --- /dev/null +++ b/src/intent-store.ts @@ -0,0 +1,233 @@ +/** + * @fileoverview Read My Mind intent store: per-case profiles of user intent. + * + * Feeds the Read My Mind predictor (`docs/readmymind-plan.md`). Each profile + * pairs user/agent-stated `goals` with the user's recently captured prompts, + * keyed by owner + realpath(workingDir) so the profile survives `/clear`, + * respawns, and session churn, and so multi-user scoping is structural (two + * owners of the same directory get distinct profiles). + * + * Capture rides the session transcript (`transcript:user_prompt`), not the + * input paths: `POST /input` sees only programmatic prompts and the WS channel + * delivers raw keystrokes, so neither yields clean submitted prompts. + * + * Prompts can contain secrets, so the state file is written 0600 (same posture + * as `users.json`) and the store is never fed into `/api/search`. + * + * Pure helpers (`deriveIntentKey`, `sanitizePromptText`, `isCapturablePrompt`, + * `appendPrompt`) are exported for unit tests; the `IntentStore` class adds the + * IO. Writes are atomic (tmp + rename) and synchronous: mutations arrive at + * human prompting pace, so there is nothing to debounce and no timer to leak. + */ + +import { createHash } from 'node:crypto'; +import { existsSync, mkdirSync, readFileSync, realpathSync, renameSync, writeFileSync } from 'node:fs'; +import { dirname } from 'node:path'; +import { dataPath } from './config/instance.js'; +import type { IntentProfile, IntentPromptEntry } from './types/index.js'; + +// ========== Limits ========== + +/** Max stored profiles; lowest `updatedAt` is evicted first. */ +export const MAX_INTENT_PROFILES = 200; + +/** Max captured prompts per profile (FIFO). */ +export const MAX_RECENT_PROMPTS = 50; + +/** Max characters kept per captured prompt. */ +export const MAX_PROMPT_CHARS = 500; + +/** Max characters for the `goals` field. */ +export const MAX_GOALS_CHARS = 8192; + +/** Prompts shorter than this are menu digits / Esc artifacts, not intent. */ +const MIN_PROMPT_CHARS = 3; + +// ========== Pure helpers ========== + +/** Stable per-case key: owner + resolved workingDir, hashed. */ +export function deriveIntentKey(owner: string | undefined, workingDir: string): string { + return createHash('sha256') + .update(`${owner ?? ''}:${workingDir}`) + .digest('hex') + .slice(0, 16); +} + +/** + * Transcript user entries that are not typed intent: local slash-command echo, + * hook/system wrappers, and interrupt markers. + */ +export function isCapturablePrompt(text: string): boolean { + if (text.includes('') || text.includes('')) return false; + if (text.startsWith('')) return false; + if (text.startsWith('Caveat: The messages below')) return false; + if (text.startsWith('[Request interrupted')) return false; + return true; +} + +/** + * Collapse a transcript prompt to a bounded single line, or null when it is + * too short to mean anything (menu digits, Esc artifacts). + */ +export function sanitizePromptText(raw: string): string | null { + const text = raw + .replace(/[\r\n]+/g, ' ') + // eslint-disable-next-line no-control-regex + .replace(/[\x00-\x08\x0b-\x1f\x7f]/g, '') + .trim(); + if (text.length < MIN_PROMPT_CHARS) return null; + return text.length > MAX_PROMPT_CHARS ? text.slice(0, MAX_PROMPT_CHARS) : text; +} + +/** + * Fold one prompt into a profile: consecutive duplicates collapse (auto-resume + * "continue" spam), FIFO cap applies. Returns a new profile object. + */ +export function appendPrompt(profile: IntentProfile, entry: IntentPromptEntry): IntentProfile { + const last = profile.recentPrompts[profile.recentPrompts.length - 1]; + if (last && last.text === entry.text) { + return { ...profile, updatedAt: entry.ts }; + } + const recentPrompts = [...profile.recentPrompts, entry].slice(-MAX_RECENT_PROMPTS); + return { ...profile, recentPrompts, updatedAt: entry.ts }; +} + +// ========== Store ========== + +interface IntentStoreFile { + version: 1; + profiles: IntentProfile[]; +} + +export class IntentStore { + private profiles: Map | null = null; + + private get filePath(): string { + return dataPath('intents.json'); + } + + // ----- Public API ----- + + /** + * The profile for a session's case. Never persists on read: an absent + * profile returns an empty transient one (`updatedAt: 0`). + */ + getProfile(owner: string | undefined, workingDir: string): IntentProfile { + const dir = this.resolveDir(workingDir); + const key = deriveIntentKey(owner, dir); + return this.load().get(key) ?? this.emptyProfile(key, dir); + } + + /** + * Capture one submitted prompt. Returns true when it was recorded (passed + * the capturability filter and sanitization). + */ + recordPrompt( + owner: string | undefined, + workingDir: string, + sessionId: string, + rawText: string, + ts: number = Date.now() + ): boolean { + if (!isCapturablePrompt(rawText)) return false; + const text = sanitizePromptText(rawText); + if (text === null) return false; + + const dir = this.resolveDir(workingDir); + const key = deriveIntentKey(owner, dir); + const profiles = this.load(); + const profile = profiles.get(key) ?? this.emptyProfile(key, dir); + profiles.set(key, appendPrompt(profile, { ts, sessionId, text })); + this.evictOverflow(profiles); + this.persist(); + return true; + } + + /** Replace the goals text (bounded). Returns the updated profile. */ + setGoals(owner: string | undefined, workingDir: string, goals: string): IntentProfile { + const dir = this.resolveDir(workingDir); + const key = deriveIntentKey(owner, dir); + const profiles = this.load(); + const profile = profiles.get(key) ?? this.emptyProfile(key, dir); + const updated: IntentProfile = { ...profile, goals: goals.slice(0, MAX_GOALS_CHARS), updatedAt: Date.now() }; + profiles.set(key, updated); + this.evictOverflow(profiles); + this.persist(); + return updated; + } + + /** Forget everything for a case. Returns true when a profile existed. */ + deleteProfile(owner: string | undefined, workingDir: string): boolean { + const dir = this.resolveDir(workingDir); + const key = deriveIntentKey(owner, dir); + const profiles = this.load(); + const existed = profiles.delete(key); + if (existed) this.persist(); + return existed; + } + + // ----- Internals ----- + + private emptyProfile(key: string, workingDir: string): IntentProfile { + return { key, workingDir, updatedAt: 0, goals: '', recentPrompts: [] }; + } + + private resolveDir(workingDir: string): string { + try { + return realpathSync(workingDir); + } catch { + return workingDir; + } + } + + private load(): Map { + if (this.profiles) return this.profiles; + this.profiles = new Map(); + try { + if (existsSync(this.filePath)) { + const parsed = JSON.parse(readFileSync(this.filePath, 'utf-8')) as IntentStoreFile; + if (parsed && Array.isArray(parsed.profiles)) { + for (const profile of parsed.profiles) { + if (profile && typeof profile.key === 'string') this.profiles.set(profile.key, profile); + } + } + } + } catch (err) { + console.warn(`[IntentStore] Failed to load ${this.filePath}, starting empty:`, err); + } + return this.profiles; + } + + private evictOverflow(profiles: Map): void { + while (profiles.size > MAX_INTENT_PROFILES) { + let oldestKey: string | null = null; + let oldestAt = Infinity; + for (const [key, profile] of profiles) { + if (profile.updatedAt < oldestAt) { + oldestAt = profile.updatedAt; + oldestKey = key; + } + } + if (oldestKey === null) return; + profiles.delete(oldestKey); + } + } + + private persist(): void { + if (!this.profiles) return; + const file: IntentStoreFile = { version: 1, profiles: [...this.profiles.values()] }; + const tmpPath = `${this.filePath}.tmp`; + try { + // dataPath()'s own mkdir is once-per-process; per-file test HOMEs need this. + mkdirSync(dirname(this.filePath), { recursive: true }); + // 0600: captured prompts can contain secrets (same posture as users.json). + writeFileSync(tmpPath, JSON.stringify(file, null, 2), { mode: 0o600 }); + renameSync(tmpPath, this.filePath); + } catch (err) { + console.warn(`[IntentStore] Failed to persist ${this.filePath}:`, err); + } + } +} + +/** Module-level singleton, same pattern as `approvalInbox` (web/approval-inbox.ts). */ +export const intentStore = new IntentStore(); diff --git a/src/transcript-watcher.ts b/src/transcript-watcher.ts index 8d44865c..0731f424 100644 --- a/src/transcript-watcher.ts +++ b/src/transcript-watcher.ts @@ -6,6 +6,7 @@ * - Tool execution state * - Error conditions * - Plan mode prompts + * - User-authored prompts (`transcript:user_prompt`, Read My Mind intent capture) * * The transcript path is provided by Claude Code hooks in the `transcript_path` field. */ @@ -372,12 +373,23 @@ export class TranscriptWatcher extends EventEmitter { this.state.errorMessage = null; const content = entry.message?.content; + if (typeof content === 'string') { + if (content.trim()) this.emit('transcript:user_prompt', content, entry.timestamp); + return; + } if (!Array.isArray(content)) return; + let promptText = ''; for (const block of content) { if (block.type === 'tool_result') { this.handleToolResult(block); + } else if (block.type === 'text' && block.text) { + promptText += (promptText ? ' ' : '') + block.text; } } + // Text blocks mean a typed prompt; tool_result-only entries are Claude's own + // tool plumbing, not intent. Filtering of command echo / system wrappers is + // the intent store's job (`isCapturablePrompt`), not the watcher's. + if (promptText.trim()) this.emit('transcript:user_prompt', promptText, entry.timestamp); } private handleToolResult(block: TranscriptContentBlock): void { diff --git a/src/types/index.ts b/src/types/index.ts index 6c810c2e..0090639a 100644 --- a/src/types/index.ts +++ b/src/types/index.ts @@ -71,3 +71,4 @@ export * from './workflow-run.js'; export * from './search.js'; export * from './user.js'; export * from './webview.js'; +export * from './intent.js'; diff --git a/src/types/intent.ts b/src/types/intent.ts new file mode 100644 index 00000000..943b9185 --- /dev/null +++ b/src/types/intent.ts @@ -0,0 +1,31 @@ +/** + * @fileoverview Read My Mind intent types. + * + * An intent profile is per CASE (owner + workingDir), not per session: + * intentions outlive `/clear`, respawn cycles, and individual sessions. + * See `docs/readmymind-plan.md`. + */ + +/** One captured user prompt, as it appeared in the session transcript. */ +export interface IntentPromptEntry { + /** Capture time (ms epoch). */ + ts: number; + /** Codeman session the prompt was sent in. */ + sessionId: string; + /** The prompt text, sanitized and bounded. */ + text: string; +} + +/** Per-case profile of what the user is trying to accomplish. */ +export interface IntentProfile { + /** Stable key: sha256(owner + ':' + realpath(workingDir)), first 16 hex chars. */ + key: string; + /** The case working directory the profile belongs to (realpath-resolved). */ + workingDir: string; + /** Last mutation (ms epoch). 0 for a never-persisted empty profile. */ + updatedAt: number; + /** User/agent-stated goals, freeform markdown, bounded. */ + goals: string; + /** Most recent captured prompts, oldest first, FIFO-capped. */ + recentPrompts: IntentPromptEntry[]; +} diff --git a/src/web/routes/index.ts b/src/web/routes/index.ts index ac57d649..a6330a6d 100644 --- a/src/web/routes/index.ts +++ b/src/web/routes/index.ts @@ -11,6 +11,7 @@ export { registerCronRoutes } from './cron-routes.js'; export { registerSystemRoutes } from './system-routes.js'; export { registerHookEventRoutes } from './hook-event-routes.js'; export { registerApprovalRoutes } from './approval-routes.js'; +export { registerReadMyMindRoutes } from './readmymind-routes.js'; export { registerStatusTelemetryRoutes } from './status-telemetry-routes.js'; export { registerCaseRoutes } from './case-routes.js'; export { registerSessionRoutes } from './session-routes.js'; diff --git a/src/web/routes/readmymind-routes.ts b/src/web/routes/readmymind-routes.ts new file mode 100644 index 00000000..8d2bd0ac --- /dev/null +++ b/src/web/routes/readmymind-routes.ts @@ -0,0 +1,50 @@ +/** + * @fileoverview Read My Mind intent routes. + * + * Per-case intent profiles feeding the Read My Mind predictor + * (docs/readmymind-plan.md): + * - `GET /api/sessions/:id/intent`: the profile for the session's case + * - `PUT /api/sessions/:id/intent`: replace the goals text + * - `DELETE /api/sessions/:id/intent`: forget the case's profile + * + * The profile is keyed by owner + workingDir, so multi-user scoping is + * structural; session ownership is still enforced via `findSessionOrFail` + * (with `req`, so a foreign session id 404s) to keep the session-routes + * no-existence-leak policy. + * + * Deliberately session-scoped rather than a raw `/api/intents/:key` surface: + * the session resolves owner + workingDir server-side, so a caller can never + * address another case's profile by guessing keys. + * + * Registrations use the bare `app.('path', ...)` + `req.params as` + * shape (session-routes style): these endpoints are documented in the agent + * skill, and the endpoints.md drift test's scanner does not see registrations + * with a generic between the method and the path. + */ + +import { FastifyInstance } from 'fastify'; +import { IntentGoalsSchema } from '../schemas.js'; +import { parseBody, findSessionOrFail } from '../route-helpers.js'; +import { intentStore } from '../../intent-store.js'; +import type { SessionPort } from '../ports/index.js'; + +export function registerReadMyMindRoutes(app: FastifyInstance, ctx: SessionPort): void { + app.get('/api/sessions/:id/intent', async (req) => { + const { id } = req.params as { id: string }; + const session = findSessionOrFail(ctx, id, req); + return { success: true, data: { intent: intentStore.getProfile(session.owner, session.workingDir) } }; + }); + + app.put('/api/sessions/:id/intent', async (req) => { + const { id } = req.params as { id: string }; + const body = parseBody(IntentGoalsSchema, req.body); + const session = findSessionOrFail(ctx, id, req); + return { success: true, data: { intent: intentStore.setGoals(session.owner, session.workingDir, body.goals) } }; + }); + + app.delete('/api/sessions/:id/intent', async (req) => { + const { id } = req.params as { id: string }; + const session = findSessionOrFail(ctx, id, req); + return { success: true, data: { deleted: intentStore.deleteProfile(session.owner, session.workingDir) } }; + }); +} diff --git a/src/web/schemas.ts b/src/web/schemas.ts index fb3eb832..f25edb98 100644 --- a/src/web/schemas.ts +++ b/src/web/schemas.ts @@ -700,6 +700,16 @@ export const ApprovalAnswerSchema = z }) .strict(); +/** + * Body of PUT /api/sessions/:id/intent (Read My Mind). The 8192 cap mirrors + * MAX_GOALS_CHARS in intent-store.ts. + */ +export const IntentGoalsSchema = z + .object({ + goals: z.string().max(8192), + }) + .strict(); + // ========== Configuration ========== /** @@ -809,6 +819,13 @@ export const SettingsUpdateSchema = z * already pending immediately. */ approvalsInboxEnabled: z.boolean().optional(), + /** + * Read My Mind (docs/readmymind-plan.md): capture the user's submitted + * prompts into per-case intent profiles. SYNCED, default OFF (opt-in: + * captured prompts are sensitive). OFF stops capture immediately; already + * stored profiles stay until DELETE /api/sessions/:id/intent. + */ + readMyMindEnabled: z.boolean().optional(), tunnelEnabled: z.boolean().optional(), // Action field (NOT persisted): explicit per-request acknowledgment that the // operator accepts exposing an UNAUTHENTICATED public tunnel (no CODEMAN_PASSWORD). diff --git a/src/web/server.ts b/src/web/server.ts index b46d06ea..a668ae51 100644 --- a/src/web/server.ts +++ b/src/web/server.ts @@ -85,7 +85,8 @@ import { attachSessionListeners, detachSessionListeners, } from './session-listener-wiring.js'; -import { sessionWaits } from './session-wait-registry.js'; +import { sessionWaits, hooksAvailableForMode } from './session-wait-registry.js'; +import { intentStore } from '../intent-store.js'; import { approvalInbox } from './approval-inbox.js'; import { wireRespawnListeners, @@ -149,6 +150,7 @@ import { registerScheduledRoutes, registerHookEventRoutes, registerApprovalRoutes, + registerReadMyMindRoutes, registerStatusTelemetryRoutes, registerSystemRoutes, registerCaseRoutes, @@ -955,6 +957,7 @@ export class WebServer extends EventEmitter { registerScheduledRoutes(this.app, ctx); registerHookEventRoutes(this.app, ctx); registerApprovalRoutes(this.app, ctx); + registerReadMyMindRoutes(this.app, ctx); registerStatusTelemetryRoutes(this.app, ctx); registerSystemRoutes(this.app, ctx); registerCaseRoutes(this.app, ctx); @@ -1022,6 +1025,10 @@ export class WebServer extends EventEmitter { console.error(`[Transcript] Error for session ${sessionId}:`, error.message); }); + watcher.on('transcript:user_prompt', (text: string) => { + void this.captureIntentPrompt(sessionId, text); + }); + this.transcriptWatchers.set(sessionId, watcher); } @@ -1029,6 +1036,24 @@ export class WebServer extends EventEmitter { watcher.updatePath(transcriptPath); } + /** + * Read My Mind intent capture: fold one transcript user prompt into the + * case's intent profile (docs/readmymind-plan.md). Opt-in via + * `readMyMindEnabled` (default OFF) and claude-only; the mode gate is + * belt-and-braces since only hook-fed sessions have a transcript watcher. + */ + private async captureIntentPrompt(sessionId: string, text: string): Promise { + const session = this.sessions.get(sessionId); + if (!session || !hooksAvailableForMode(session.mode)) return; + try { + const settings = await this.readSettings(); + if (settings.readMyMindEnabled !== true) return; + intentStore.recordPrompt(session.owner, session.workingDir, sessionId, text); + } catch (err) { + console.warn(`[IntentStore] Capture failed for session ${sessionId}:`, err); + } + } + /** * Stop the transcript watcher for a session. */ diff --git a/test/intent-store.test.ts b/test/intent-store.test.ts new file mode 100644 index 00000000..3ad1a354 --- /dev/null +++ b/test/intent-store.test.ts @@ -0,0 +1,187 @@ +/** + * @fileoverview Unit tests for the Read My Mind intent store (src/intent-store.ts). + * + * Pure helpers (key derivation, capturability filter, sanitization, append fold) + * plus the IO layer against a per-test temp data dir (CODEMAN_DATA_DIR) so + * nothing touches the real ~/.codeman. No server, no tmux. + */ + +import { afterEach, beforeEach, describe, expect, it } from 'vitest'; +import fs from 'node:fs/promises'; +import { statSync, existsSync } from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; + +import { + appendPrompt, + deriveIntentKey, + IntentStore, + isCapturablePrompt, + MAX_GOALS_CHARS, + MAX_INTENT_PROFILES, + MAX_PROMPT_CHARS, + MAX_RECENT_PROMPTS, + sanitizePromptText, +} from '../src/intent-store.js'; +import type { IntentProfile } from '../src/types/index.js'; + +let tmpDir: string; +let savedDataDir: string | undefined; + +beforeEach(async () => { + tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'codeman-intents-')); + savedDataDir = process.env.CODEMAN_DATA_DIR; + process.env.CODEMAN_DATA_DIR = tmpDir; +}); + +afterEach(async () => { + if (savedDataDir === undefined) delete process.env.CODEMAN_DATA_DIR; + else process.env.CODEMAN_DATA_DIR = savedDataDir; + await fs.rm(tmpDir, { recursive: true, force: true }); +}); + +const intentsFile = () => path.join(tmpDir, 'intents.json'); + +function makeProfile(overrides: Partial = {}): IntentProfile { + return { key: 'k', workingDir: '/w', updatedAt: 0, goals: '', recentPrompts: [], ...overrides }; +} + +describe('deriveIntentKey', () => { + it('is stable and 16 lowercase hex chars', () => { + const a = deriveIntentKey('alice', '/home/alice/proj'); + expect(a).toMatch(/^[0-9a-f]{16}$/); + expect(deriveIntentKey('alice', '/home/alice/proj')).toBe(a); + }); + + it('separates owners and directories', () => { + expect(deriveIntentKey('alice', '/p')).not.toBe(deriveIntentKey('bob', '/p')); + expect(deriveIntentKey('alice', '/p')).not.toBe(deriveIntentKey('alice', '/q')); + expect(deriveIntentKey(undefined, '/p')).not.toBe(deriveIntentKey('alice', '/p')); + }); +}); + +describe('isCapturablePrompt', () => { + it('rejects local command echo and system wrappers', () => { + expect(isCapturablePrompt('/model')).toBe(false); + expect(isCapturablePrompt('before out')).toBe(false); + expect(isCapturablePrompt('context')).toBe(false); + expect(isCapturablePrompt('Caveat: The messages below were generated…')).toBe(false); + expect(isCapturablePrompt('[Request interrupted by user]')).toBe(false); + }); + + it('accepts a normal prompt', () => { + expect(isCapturablePrompt('fix the login bug and add a test')).toBe(true); + }); +}); + +describe('sanitizePromptText', () => { + it('collapses newlines and strips control chars', () => { + expect(sanitizePromptText('line one\nline two\r\nthree')).toBe('line one line two three'); + expect(sanitizePromptText('a\x1b[31mred\x1b[0mb end')).toBe('a[31mred[0mb end'); + }); + + it('returns null for menu-digit noise', () => { + expect(sanitizePromptText('1')).toBeNull(); + expect(sanitizePromptText(' \n ')).toBeNull(); + }); + + it('truncates to the cap', () => { + const out = sanitizePromptText('x'.repeat(MAX_PROMPT_CHARS + 100)); + expect(out).toHaveLength(MAX_PROMPT_CHARS); + }); +}); + +describe('appendPrompt', () => { + it('collapses consecutive duplicates but keeps non-adjacent ones', () => { + let p = makeProfile(); + p = appendPrompt(p, { ts: 1, sessionId: 's', text: 'continue' }); + p = appendPrompt(p, { ts: 2, sessionId: 's', text: 'continue' }); + expect(p.recentPrompts).toHaveLength(1); + expect(p.updatedAt).toBe(2); + p = appendPrompt(p, { ts: 3, sessionId: 's', text: 'run tests' }); + p = appendPrompt(p, { ts: 4, sessionId: 's', text: 'continue' }); + expect(p.recentPrompts.map((e) => e.text)).toEqual(['continue', 'run tests', 'continue']); + }); + + it('FIFO-caps at MAX_RECENT_PROMPTS, dropping the oldest', () => { + let p = makeProfile(); + for (let i = 0; i < MAX_RECENT_PROMPTS + 5; i++) { + p = appendPrompt(p, { ts: i, sessionId: 's', text: `prompt number ${i}` }); + } + expect(p.recentPrompts).toHaveLength(MAX_RECENT_PROMPTS); + expect(p.recentPrompts[0].text).toBe('prompt number 5'); + }); +}); + +describe('IntentStore', () => { + it('records a prompt, persists 0600, and reloads from disk', () => { + const store = new IntentStore(); + expect(store.recordPrompt('alice', tmpDir, 'sess1', 'ship the release')).toBe(true); + expect(existsSync(intentsFile())).toBe(true); + expect(statSync(intentsFile()).mode & 0o777).toBe(0o600); + + const reloaded = new IntentStore(); + const profile = reloaded.getProfile('alice', tmpDir); + expect(profile.recentPrompts.map((e) => e.text)).toEqual(['ship the release']); + expect(profile.updatedAt).toBeGreaterThan(0); + }); + + it('getProfile on an absent case returns an empty transient profile without persisting', () => { + const store = new IntentStore(); + const profile = store.getProfile('alice', tmpDir); + expect(profile.updatedAt).toBe(0); + expect(profile.goals).toBe(''); + expect(profile.recentPrompts).toEqual([]); + expect(existsSync(intentsFile())).toBe(false); + }); + + it('filters uncapturable and too-short prompts', () => { + const store = new IntentStore(); + expect(store.recordPrompt('a', tmpDir, 's', '/clear')).toBe(false); + expect(store.recordPrompt('a', tmpDir, 's', '2')).toBe(false); + expect(existsSync(intentsFile())).toBe(false); + }); + + it('keys by resolved directory so path spellings converge', () => { + const store = new IntentStore(); + store.recordPrompt('a', `${tmpDir}${path.sep}.`, 's', 'same case either way'); + const profile = store.getProfile('a', tmpDir); + expect(profile.recentPrompts).toHaveLength(1); + }); + + it('separates owners of the same directory', () => { + const store = new IntentStore(); + store.recordPrompt('alice', tmpDir, 's', 'alice private plan'); + expect(store.getProfile('bob', tmpDir).recentPrompts).toEqual([]); + }); + + it('setGoals bounds the text and deleteProfile forgets the case', () => { + const store = new IntentStore(); + const updated = store.setGoals('a', tmpDir, 'g'.repeat(MAX_GOALS_CHARS + 50)); + expect(updated.goals).toHaveLength(MAX_GOALS_CHARS); + + expect(store.deleteProfile('a', tmpDir)).toBe(true); + expect(store.deleteProfile('a', tmpDir)).toBe(false); + expect(store.getProfile('a', tmpDir).goals).toBe(''); + }); + + it('evicts the least-recently-updated profile past the cap', () => { + const store = new IntentStore(); + for (let i = 0; i <= MAX_INTENT_PROFILES; i++) { + store.setGoals('a', `${tmpDir}/case-${i}`, `goal ${i}`); + } + const reloaded = new IntentStore(); + expect(reloaded.getProfile('a', `${tmpDir}/case-0`).goals).toBe(''); + expect(reloaded.getProfile('a', `${tmpDir}/case-${MAX_INTENT_PROFILES}`).goals).toBe(`goal ${MAX_INTENT_PROFILES}`); + }); + + it('starts empty on a corrupted state file', () => { + const store = new IntentStore(); + store.setGoals('a', tmpDir, 'valid'); + return fs.writeFile(intentsFile(), '{ not json').then(() => { + const reloaded = new IntentStore(); + expect(reloaded.getProfile('a', tmpDir).goals).toBe(''); + expect(reloaded.recordPrompt('a', tmpDir, 's', 'recover cleanly')).toBe(true); + }); + }); +}); diff --git a/test/routes/readmymind-routes.test.ts b/test/routes/readmymind-routes.test.ts new file mode 100644 index 00000000..4aa1c169 --- /dev/null +++ b/test/routes/readmymind-routes.test.ts @@ -0,0 +1,148 @@ +/** + * @fileoverview Read My Mind intent route tests (src/web/routes/readmymind-routes.ts) + * via app.inject(), no live port. + * + * The routes read the process-wide `intentStore` singleton, whose data file + * resolves under this test file's temp HOME (test/setup.ts). The singleton's + * in-memory map lives for the whole file, so each test uses a distinct + * session workingDir to stay isolated. + * + * Port: SessionPort. + */ +import { describe, it, expect, beforeEach, afterEach } from 'vitest'; +import { registerReadMyMindRoutes } from '../../src/web/routes/readmymind-routes.js'; +import { createRouteTestHarness, type RouteTestHarness } from './_route-test-utils.js'; + +const SESSION_ID = 'test-session-1'; + +let harness: RouteTestHarness; +let caseCounter = 0; + +beforeEach(async () => { + harness = await createRouteTestHarness(registerReadMyMindRoutes); + // Unique (nonexistent) workingDir per test: resolveDir falls back to the raw + // string, so the key is stable and no other test's profile bleeds in. + caseCounter++; + sessionUnderTest().workingDir = `/nonexistent/readmymind-case-${caseCounter}`; +}); + +afterEach(async () => { + await harness.app.close(); +}); + +function sessionUnderTest(): { workingDir: string; owner?: string } { + return harness.ctx.sessions.get(SESSION_ID) as unknown as { workingDir: string; owner?: string }; +} + +describe('GET /api/sessions/:id/intent', () => { + it('returns an empty transient profile for a fresh case', async () => { + const res = await harness.app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/intent` }); + expect(res.statusCode).toBe(200); + const body = res.json(); + expect(body.success).toBe(true); + expect(body.data.intent.goals).toBe(''); + expect(body.data.intent.recentPrompts).toEqual([]); + expect(body.data.intent.updatedAt).toBe(0); + }); + + it('404s an unknown session id', async () => { + const res = await harness.app.inject({ method: 'GET', url: '/api/sessions/nope/intent' }); + expect(res.statusCode).toBe(404); + expect(res.json().success).toBe(false); + }); +}); + +describe('PUT /api/sessions/:id/intent', () => { + it('round-trips goals through the store', async () => { + const put = await harness.app.inject({ + method: 'PUT', + url: `/api/sessions/${SESSION_ID}/intent`, + payload: { goals: 'ship 1.17 with the readmymind phase 1' }, + }); + expect(put.statusCode).toBe(200); + expect(put.json().data.intent.goals).toBe('ship 1.17 with the readmymind phase 1'); + expect(put.json().data.intent.updatedAt).toBeGreaterThan(0); + + const get = await harness.app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/intent` }); + expect(get.json().data.intent.goals).toBe('ship 1.17 with the readmymind phase 1'); + }); + + it('rejects over-long goals and unknown keys (strict schema)', async () => { + const tooLong = await harness.app.inject({ + method: 'PUT', + url: `/api/sessions/${SESSION_ID}/intent`, + payload: { goals: 'x'.repeat(8193) }, + }); + expect(tooLong.statusCode).toBe(400); + + const extraKey = await harness.app.inject({ + method: 'PUT', + url: `/api/sessions/${SESSION_ID}/intent`, + payload: { goals: 'ok', recentPrompts: [] }, + }); + expect(extraKey.statusCode).toBe(400); + }); +}); + +describe('DELETE /api/sessions/:id/intent', () => { + it('forgets the case and reports whether anything existed', async () => { + await harness.app.inject({ + method: 'PUT', + url: `/api/sessions/${SESSION_ID}/intent`, + payload: { goals: 'temporary' }, + }); + + const first = await harness.app.inject({ method: 'DELETE', url: `/api/sessions/${SESSION_ID}/intent` }); + expect(first.statusCode).toBe(200); + expect(first.json().data.deleted).toBe(true); + + const second = await harness.app.inject({ method: 'DELETE', url: `/api/sessions/${SESSION_ID}/intent` }); + expect(second.json().data.deleted).toBe(false); + + const get = await harness.app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/intent` }); + expect(get.json().data.intent.goals).toBe(''); + }); +}); + +describe('multi-user scoping', () => { + let savedMultiuser: string | undefined; + + beforeEach(() => { + savedMultiuser = process.env.CODEMAN_MULTIUSER; + process.env.CODEMAN_MULTIUSER = '1'; + }); + + afterEach(() => { + if (savedMultiuser === undefined) delete process.env.CODEMAN_MULTIUSER; + else process.env.CODEMAN_MULTIUSER = savedMultiuser; + }); + + it("404s (never 403s) another user's session", async () => { + const scoped = await createRouteTestHarness(registerReadMyMindRoutes, { + authUser: { username: 'bob', role: 'user' }, + }); + try { + (scoped.ctx.sessions.get(SESSION_ID) as unknown as { owner?: string }).owner = 'alice'; + const res = await scoped.app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/intent` }); + expect(res.statusCode).toBe(404); + } finally { + await scoped.app.close(); + } + }); + + it('serves the owner normally', async () => { + const scoped = await createRouteTestHarness(registerReadMyMindRoutes, { + authUser: { username: 'bob', role: 'user' }, + }); + try { + const session = scoped.ctx.sessions.get(SESSION_ID) as unknown as { owner?: string; workingDir: string }; + session.owner = 'bob'; + session.workingDir = `/nonexistent/readmymind-owned-${Date.now()}`; + const res = await scoped.app.inject({ method: 'GET', url: `/api/sessions/${SESSION_ID}/intent` }); + expect(res.statusCode).toBe(200); + expect(res.json().success).toBe(true); + } finally { + await scoped.app.close(); + } + }); +}); diff --git a/test/transcript-watcher.test.ts b/test/transcript-watcher.test.ts index badba35e..24234d8f 100644 --- a/test/transcript-watcher.test.ts +++ b/test/transcript-watcher.test.ts @@ -232,6 +232,88 @@ describe('TranscriptWatcher', () => { }); }); + describe('User prompt capture (Read My Mind)', () => { + it('emits transcript:user_prompt with the raw text for string content', async () => { + writeFileSync(testFile, ''); + watcher.start(testFile); + + const promptHandler = vi.fn(); + watcher.on('transcript:user_prompt', promptHandler); + + const ts = new Date().toISOString(); + appendFileSync( + testFile, + JSON.stringify({ type: 'user', timestamp: ts, message: { role: 'user', content: 'fix the login bug' } }) + '\n' + ); + + await vi.waitFor(() => { + expect(promptHandler).toHaveBeenCalledWith('fix the login bug', ts); + }); + }); + + it('emits joined text blocks but stays silent for tool_result-only entries', async () => { + writeFileSync(testFile, ''); + watcher.start(testFile); + + const promptHandler = vi.fn(); + watcher.on('transcript:user_prompt', promptHandler); + + appendFileSync( + testFile, + JSON.stringify({ + type: 'user', + timestamp: new Date().toISOString(), + message: { + role: 'user', + content: [{ type: 'tool_result', tool_use_id: 'toolu_1', content: 'ok', is_error: false }], + }, + }) + '\n' + ); + appendFileSync( + testFile, + JSON.stringify({ + type: 'user', + timestamp: new Date().toISOString(), + message: { + role: 'user', + content: [ + { type: 'text', text: 'run the tests' }, + { type: 'text', text: 'then push' }, + ], + }, + }) + '\n' + ); + + await vi.waitFor(() => { + expect(promptHandler).toHaveBeenCalledTimes(1); + }); + expect(promptHandler).toHaveBeenCalledWith('run the tests then push', expect.any(String)); + }); + + it('does not emit for whitespace-only string content', async () => { + writeFileSync(testFile, ''); + watcher.start(testFile); + + const promptHandler = vi.fn(); + watcher.on('transcript:user_prompt', promptHandler); + + appendFileSync( + testFile, + JSON.stringify({ + type: 'user', + timestamp: new Date().toISOString(), + message: { role: 'user', content: ' ' }, + }) + '\n' + ); + + // Wait for the entry to be processed, then assert no emission happened. + await vi.waitFor(() => { + expect(watcher.getState().entryCount).toBeGreaterThanOrEqual(1); + }); + expect(promptHandler).not.toHaveBeenCalled(); + }); + }); + describe('State Management', () => { it('should return a copy of state', () => { const state1 = watcher.getState();