Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
695e4047a1 | ||
|
|
f475caab87 | ||
|
|
8453e953fd | ||
|
|
a8e0e2a343 | ||
|
|
4d0586a2aa | ||
|
|
67a15b5949 | ||
|
|
6bf69a82c8 | ||
|
|
d2efaa255b | ||
|
|
a721af4552 | ||
|
|
e6b18fd126 | ||
|
|
6ee88be549 | ||
|
|
1316725fdc | ||
|
|
187ce653ae | ||
|
|
da00fa6038 | ||
|
|
a36543c1b9 | ||
|
|
dea015dc91 | ||
|
|
333dc047c3 | ||
|
|
a51c17170e | ||
|
|
6ea73a9251 | ||
|
|
20c01d5b11 | ||
|
|
ce65d5f2ad | ||
|
|
cc45191c62 | ||
|
|
d897c9a1cf | ||
|
|
880b63d2a0 | ||
|
|
eb874339dd | ||
|
|
29d3fd48c1 | ||
|
|
cf6fabc070 | ||
|
|
62b7c4903b | ||
|
|
94b26f7606 | ||
|
|
ef01fb35b3 | ||
|
|
b5ea7112a9 | ||
|
|
95b00357b6 | ||
|
|
59145c48fc | ||
|
|
8dc850f845 | ||
|
|
eea84db05e | ||
|
|
ceca85365c | ||
|
|
44439c951b | ||
|
|
b2f8b03b3c | ||
|
|
afea6d6a1c | ||
|
|
2e341e3897 | ||
|
|
5459da5f9d | ||
|
|
b00a680d42 | ||
|
|
e3c496e1a4 | ||
|
|
eb831487a0 | ||
|
|
68594ac395 | ||
|
|
ec38fd11bf | ||
|
|
06f9ff6d9c | ||
|
|
257695ff8e | ||
|
|
2cfccc745f | ||
|
|
016c23934f | ||
|
|
896dc5b177 | ||
|
|
196646a7ff | ||
|
|
1b652ceb87 | ||
|
|
8abf349cfc | ||
|
|
ae5bcf9330 | ||
|
|
78d5fcf70c | ||
|
|
1ff315a1e6 | ||
|
|
08de6667ab | ||
|
|
d27f8e77f7 | ||
|
|
e248cd8bcf | ||
|
|
73d81afd4d | ||
|
|
7884a37c55 | ||
|
|
ad89a97106 | ||
|
|
0600b7843e | ||
|
|
6d896c781e | ||
|
|
930492058b | ||
|
|
101cee0cec | ||
|
|
7752325c90 | ||
|
|
6b284598cf | ||
|
|
94bcf524a2 | ||
|
|
98966def03 | ||
|
|
e87b03b6c2 | ||
|
|
edd494ec5f | ||
|
|
00721069e1 | ||
|
|
453a5383d2 | ||
|
|
e7b95ae579 | ||
|
|
56c2c29009 | ||
|
|
7beec7194a | ||
|
|
dcc814f40c | ||
|
|
b7e94e7068 | ||
|
|
eade261763 | ||
|
|
41a82fcf02 | ||
|
|
eecf74c001 | ||
|
|
23b4dfcd82 | ||
|
|
e017b275fe | ||
|
|
e8a809ea80 | ||
|
|
8006cc5db3 | ||
|
|
0ded279b55 | ||
|
|
aa5724c390 | ||
|
|
f21df2a9fb | ||
|
|
e549e15cb8 | ||
|
|
d07b59db4e | ||
|
|
a5a7e0c94c | ||
|
|
79d7117e6d | ||
|
|
3cf486730b | ||
|
|
996b096849 | ||
|
|
ffa7fcf839 | ||
|
|
a1c69f7405 | ||
|
|
534899bc2b | ||
|
|
03d91ffddd | ||
|
|
6280998bd8 | ||
|
|
02e2f3e8b5 | ||
|
|
41300f0a34 | ||
|
|
adbc083426 | ||
|
|
3754bcd1aa | ||
|
|
f2f909ca9c | ||
|
|
1c3f2f6571 | ||
|
|
8da1bdf690 | ||
|
|
a93325b312 | ||
|
|
2d03e4efc9 | ||
|
|
ab7c502c2a | ||
|
|
546bbcbe7c | ||
|
|
774d5ff321 | ||
|
|
98ceb5da1d | ||
|
|
34fb5e49f8 | ||
|
|
002cf81b1e | ||
|
|
9b4aab2502 | ||
|
|
829c797726 | ||
|
|
85da3bb898 | ||
|
|
29b2653801 | ||
|
|
7b8b175133 | ||
|
|
c3027b21e1 | ||
|
|
0b231edd43 | ||
|
|
ea1c2ee4ec | ||
|
|
b4a808adcf | ||
|
|
f3cbe9bca6 | ||
|
|
a11bcb0029 | ||
|
|
47fd9a922f | ||
|
|
14f7d8298d | ||
|
|
d32f4debb2 | ||
|
|
3cb7b510f8 | ||
|
|
6a12a72c9c | ||
|
|
f1a126efeb | ||
|
|
12fd780af8 | ||
|
|
fd74a42933 | ||
|
|
7101e64800 | ||
|
|
1b10d9b733 | ||
|
|
5078f5251d | ||
|
|
196af8fba7 | ||
|
|
a9b22b86a4 | ||
|
|
28cace5858 | ||
|
|
0a594b61bd | ||
|
|
89d787a949 | ||
|
|
bd9797b68c | ||
|
|
0e6cd94312 | ||
|
|
24a6f1cac8 | ||
|
|
8e679a280b | ||
|
|
c642689bbd | ||
|
|
28a6247c27 | ||
|
|
0ceb455c4b | ||
|
|
e51117dfa9 | ||
|
|
13d41cf7c7 | ||
|
|
2c7557d002 | ||
|
|
53b473708f | ||
|
|
2cba393ae5 | ||
|
|
64b8ea30b2 | ||
|
|
5743af3339 | ||
|
|
2011bd8d89 | ||
|
|
b76724690d | ||
|
|
f277f9664c | ||
|
|
cd49171bbc | ||
|
|
0f57342b10 | ||
|
|
e1f0ac993a | ||
|
|
a84ef52992 | ||
|
|
692c894760 | ||
|
|
ad0acb6d58 | ||
|
|
d866c8f30e | ||
|
|
28537de39d | ||
|
|
ba09184efa | ||
|
|
93719b41cd | ||
|
|
a448983be3 | ||
|
|
3145eac6d9 | ||
|
|
e3c609f5f0 | ||
|
|
52e774f83c | ||
|
|
82d08df53f | ||
|
|
2709b2fe49 | ||
|
|
b1d3b27e5b | ||
|
|
47963b54fa | ||
|
|
b7c3c30c8c | ||
|
|
0d80524f10 | ||
|
|
a9d83ec4e3 | ||
|
|
84137cdba4 | ||
|
|
6eb3969816 | ||
|
|
de49437a6f | ||
|
|
6a27639083 | ||
|
|
eb1b38c718 | ||
|
|
ea7b103b47 | ||
|
|
bec8e2f9ee | ||
|
|
2203f3a347 | ||
|
|
867a10d78a | ||
|
|
e54d7badc4 | ||
|
|
40dfac3534 | ||
|
|
e899a43a18 | ||
|
|
0cab8a7ece | ||
|
|
7b7cf958c0 | ||
|
|
6e64ddd853 | ||
|
|
afea91b92b | ||
|
|
9449a8f157 | ||
|
|
d322f17f73 | ||
|
|
61b5ec095c | ||
|
|
497ca4891a | ||
|
|
580b7a3f90 | ||
|
|
34c3d8f5ff | ||
|
|
2491471ba5 | ||
|
|
0c4aac8029 | ||
|
|
d436c6375f | ||
|
|
e96baf9f66 |
@@ -22,6 +22,9 @@ jobs:
|
||||
- name: Install dependencies
|
||||
run: npm ci
|
||||
|
||||
- name: Check package-lock.json version sync
|
||||
run: npm run check:lockfile
|
||||
|
||||
- name: Type check
|
||||
run: npm run typecheck
|
||||
|
||||
@@ -31,6 +34,32 @@ jobs:
|
||||
- name: Format check
|
||||
run: npm run format:check
|
||||
|
||||
- name: Server boot smoke test
|
||||
run: |
|
||||
set -u
|
||||
if ! command -v tmux >/dev/null; then
|
||||
sudo apt-get update -qq
|
||||
sudo apt-get install -y tmux
|
||||
fi
|
||||
npx tsx src/index.ts web --port 3151 > /tmp/boot.log 2>&1 &
|
||||
SERVER_PID=$!
|
||||
trap "kill $SERVER_PID 2>/dev/null || true" EXIT
|
||||
for i in $(seq 1 30); do
|
||||
if curl -fsS http://localhost:3151/api/status -o /dev/null; then
|
||||
echo "Server booted in ${i}s"
|
||||
exit 0
|
||||
fi
|
||||
if ! kill -0 $SERVER_PID 2>/dev/null; then
|
||||
echo "Server exited before becoming ready. Logs:"
|
||||
cat /tmp/boot.log
|
||||
exit 1
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
echo "Server did not respond on /api/status within 30s. Logs:"
|
||||
cat /tmp/boot.log
|
||||
exit 1
|
||||
|
||||
# Note: The test suite is intentionally excluded from CI.
|
||||
# Tests spawn real tmux sessions and require a full system environment.
|
||||
# Run tests locally with: npx vitest run test/<file>.test.ts
|
||||
|
||||
@@ -48,7 +48,22 @@ Thumbs.db
|
||||
# Generated output
|
||||
out/
|
||||
screenshots-echo-diag/
|
||||
tools/remotion/out/
|
||||
scripts/remotion/out/
|
||||
|
||||
# Artifacts that should not be tracked
|
||||
test-results/
|
||||
tmp/
|
||||
# Root `public` (a symlink to scripts/remotion/public — local artifact). ANCHORED
|
||||
# with a leading slash so it does NOT also match src/web/public (a bare `public`
|
||||
# would swallow the whole web UI source dir and silently un-stage any new asset
|
||||
# added there). No trailing slash so it still matches the symlink, not just dirs.
|
||||
/public
|
||||
|
||||
# Opt-in gesture overlay runtime assets: large MediaPipe wasm + model (~27 MB)
|
||||
# fetched at build/install by scripts/fetch-gesture-assets.mjs, kept out of git.
|
||||
# (The gesture bundle itself, gesture-codeman.js, IS tracked.)
|
||||
src/web/public/gesture/wasm/
|
||||
src/web/public/gesture/*.task
|
||||
|
||||
# Claude Code plan tracking
|
||||
plan.json
|
||||
@@ -61,3 +76,6 @@ commands
|
||||
todo.md
|
||||
@fix_plan.md
|
||||
readme-preview.mjs
|
||||
|
||||
# Uploaded images land here under each session working dir (runtime artifact)
|
||||
.claude-images/
|
||||
|
||||
@@ -2,8 +2,27 @@ dist/
|
||||
coverage/
|
||||
node_modules/
|
||||
src/web/public/vendor/
|
||||
src/web/public/gesture/
|
||||
src/web/public/app.js
|
||||
src/web/public/styles.css
|
||||
src/web/public/mobile.css
|
||||
src/web/public/index.html
|
||||
tools/
|
||||
# Hand-formatted public JS modules (never prettier-enforced; the new
|
||||
# check-public-assets.mjs still validates NUL bytes + JS syntax on these).
|
||||
src/web/public/constants.js
|
||||
src/web/public/image-input.js
|
||||
src/web/public/input-cjk.js
|
||||
src/web/public/keyboard-accessory.js
|
||||
src/web/public/notification-manager.js
|
||||
src/web/public/orchestrator-panel.js
|
||||
src/web/public/panels-ui.js
|
||||
src/web/public/ralph-panel.js
|
||||
src/web/public/ralph-wizard.js
|
||||
src/web/public/respawn-ui.js
|
||||
src/web/public/session-ui.js
|
||||
src/web/public/settings-ui.js
|
||||
src/web/public/sw.js
|
||||
src/web/public/terminal-ui.js
|
||||
src/web/public/voice-input.js
|
||||
src/web/public/upload.html
|
||||
scripts/remotion/
|
||||
|
||||
@@ -1,5 +1,464 @@
|
||||
# aicodeman
|
||||
|
||||
## 0.9.1
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Multi-monitor & settings UX fixes.
|
||||
- **Multi-monitor button (remote servers):** the "span displays" button spawns `scripts/span-codeman.sh` server-side, so on a non-macOS Codeman server it can't open a window on your machine. The non-macOS API error now explains this and points to running the script locally on your Mac with the remote server URL; the script header documents the same remote-client workflow.
|
||||
- **App Settings modal:** stop the modal overflowing horizontally on narrow viewports.
|
||||
- **systemd:** sync the `codeman-web.service` template with the deployed unit.
|
||||
|
||||
## 0.9.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- Security hardening release: network-bind policy, auth lockout recovery, download/SVG hardening, dependency & supply-chain fixes, tmux launch reliability, and a full security-architecture doc.
|
||||
|
||||
**Network binding (COD-29, #107):**
|
||||
- The web server now defaults to binding `127.0.0.1` (loopback) instead of `0.0.0.0`, so a fresh install is reachable only from the same machine and needs no password. New `--host` / `-H` / `CODEMAN_HOST` flag to choose the bind host.
|
||||
- Binding a non-loopback host **without** `CODEMAN_PASSWORD` no longer refuses to start — it **starts and prints a loud warning** with the three ways to secure it (set `CODEMAN_PASSWORD`, bind loopback + an authenticated tunnel / `tailscale serve`, or acknowledge with `--allow-unauthenticated-network` / `CODEMAN_ALLOW_UNAUTHENTICATED_NETWORK=1`). This keeps Codeman "just working" for new users while making remote exposure a guided, explicit choice. Host classification lives in the new `src/web/network-auth-policy.ts` (handles `127.0.0.0/8`, `::1`, `::ffff:127.*`, bracketed IPv6).
|
||||
- A post-install security note now explains the loopback default and how to expose safely.
|
||||
|
||||
**Authentication (COD-29, #107):**
|
||||
- Auth lockout now recovers gracefully: the per-IP rate-limit (`429`) check runs **after** the cookie/credential checks, so a valid session cookie or correct password is never locked out by a prior attacker's failures from the same IP (important behind a shared-IP tunnel). Wrong credentials are still counted and still hit the limit, and a `Retry-After` header is returned.
|
||||
|
||||
**Downloads & content-type hardening (COD-29, #107):**
|
||||
- New session-scoped `POST /api/download` route: realpath-bounded to the session working dir, a sensitive-path blocklist (`/etc/shadow`, `~/.ssh/`, `.env`, `*credentials*`, …), `isFile()` + 50 MB cap, forced `attachment`.
|
||||
- Workspace `.svg` files are served as `application/octet-stream` + `attachment` + `nosniff` (closes a stored-XSS-via-SVG vector); `nosniff` now applies to all `file-raw` responses.
|
||||
|
||||
**Dependencies & supply chain (COD-28, #106):**
|
||||
- Bumped security-sensitive deps to patched versions (`@fastify/static` 9, `fastify` 5.8, `uuid` 14, `vitest` 4.1, …) and added `overrides` for patched transitives (`picomatch`, `basic-ftp`, `fast-uri`, `flatted`); `npm audit` goes from 7 advisories to 0.
|
||||
- New `npm run check:public-assets` (`scripts/check-public-assets.mjs`): scans `src/web/public/**` for literal NUL bytes and runs `node --check` on every `.js` file, plus a Prettier pass on maintained files. Removed literal NUL placeholders from `app.js`. Added `test/dependency-security.test.ts` and `test/frontend-public-tooling.test.ts`.
|
||||
|
||||
**tmux launch reliability (COD-31, #110):**
|
||||
- New tmux sessions and respawns launch from a stable `/tmp` and `cd` into the workspace inside the pane, avoiding `new-session` crashes when a FUSE/rclone-mounted workspace has a transient mount blip at launch. The `cd "<dir>" && <cmd>` form is fail-safe (the CLI never runs in `/tmp`) and the path is validated + double-quoted.
|
||||
|
||||
**Test stability (COD-30, #108):**
|
||||
- Cleared leaked auth env in the Vitest setup, corrected stale route status-code / SSE-lifecycle expectations to match shipped behavior, updated the mobile keyboard accessory expectations, and measured DOMContentLoaded via browser navigation timing. Also fixed the `WebServer` title tests for the new `host` constructor arg + async `renderIndexHtml`.
|
||||
|
||||
**Docs:**
|
||||
- New `docs/security-architecture.md` documenting the full model (network binding, auth pipeline, the tunnel `req.ip` caveat, file-serving hardening, supply-chain, multi-instance isolation, security headers, and recommended secure setups). CLAUDE.md updated accordingly.
|
||||
|
||||
## 0.8.2
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Session detach/undock, opt-in gesture-control overlay, multi-monitor spanning, new App-Settings toggles, and asset cache-busting.
|
||||
- **Session detach/undock + instance isolation (#103):** Detach a session into its own solo (popup) window from the tab strip. Adds multi-instance isolation primitives in `src/config/instance.ts` (`getDataDir()`/`dataPath()`/`DEFAULT_TMUX_SOCKET`) keyed off `CODEMAN_INSTANCE`, so a beta can run side-by-side with prod without discovering/attaching to prod's live tmux sessions or clobbering its `state.json`. `CODEMAN_INSTANCE` defaults to the production layout (`~/.codeman`, `-L codeman`, port 3000), so master installs are unaffected. Adds `scripts/run-beta.sh` (`CODEMAN_INSTANCE=beta` + `CODEMAN_PORT=5000`). The legacy `~/.claudeman` migration is now scoped to the default instance only. Hardened detach edge cases. Tests: `test/config/instance.test.ts`.
|
||||
- **Gesture-control overlay (Phase 5, opt-in via `CODEMAN_GESTURE=1`):** Camera hand-tracking overlay (self-hosted MediaPipe — wasm + model fetched at install/build via `scripts/fetch-gesture-assets.mjs` rather than committed). `CODEMAN_GESTURE=1` makes the feature _available_ (CSP widening + `/gesture/` assets + `window.__codemanGestureAvailable`); the per-user **Gesture Control (beta)** toggle (App Settings → Display → Input, default OFF) is the actual on/off and reloads the page to inject/remove the bundle. Dashboard-only (not solo popups). Labeled "(beta)" (#109).
|
||||
- **Multi-monitor button:** Header button (opt-in via App Settings → Display → Header Displays) that POSTs `/api/system/span-displays` to spawn `scripts/span-codeman.sh` — a maximized browser `--app` window sized to the union of all displays, so the gesture layer's floating panels can drag across the physical monitor seam. Tests: `test/routes/system-span-displays.test.ts`.
|
||||
- **New App-Settings toggles (#105):** Gesture control and the multi-monitor button are both opt-in (default OFF), with live show/hide on save.
|
||||
- **Asset cache-busting:** `renderIndexHtml` appends `?v=<mtime>` to every same-origin `.js`/`.css` reference; `index.html` is served `no-cache`, so a normal reload picks up edited modules/styles without a hard refresh. Tests: `test/render-index-html.test.ts`.
|
||||
- **Gesture Control toggle placement:** the toggle now lives inside the existing **Input** settings section (alongside Local Echo / CJK Input / Extended Keyboard Bar) instead of a duplicate "Input" section; only the toggle itself is hidden when `CODEMAN_GESTURE=1` is unset, leaving the rest of the section intact.
|
||||
- **Service env:** `scripts/codeman-web.service` now sets `CODEMAN_GESTURE=1` so the gesture feature is available on the local install (still gated behind the default-OFF per-user toggle).
|
||||
- **Docs:** CLAUDE.md updated for the orchestrator loop, multi-monitor/span-displays, cache-busting, gesture/multi-monitor toggles, and structural-count fixes.
|
||||
|
||||
## 0.8.1
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Thinking Effort now flows as a soft default the user can override in-session (PR #104, by @TeigenZhang).
|
||||
|
||||
Previously Codeman carried the effort setting as the `CLAUDE_CODE_EFFORT_LEVEL` env var, which Claude Code treats as a hard override — it locked effort for the whole session and rejected in-session `/effort` switching (including switching to `ultracode`). Effort is now injected at spawn time as a CLI soft default that `/effort` can still change freely in either direction:
|
||||
- Regular levels (`low`/`medium`/`high`/`xhigh`/`max`) are passed via `claude --effort <level>` (the settings `effortLevel` key silently drops `max`, so the flag is used instead).
|
||||
- `ultracode` (xhigh effort + standing dynamic-workflow orchestration) is passed via `claude --settings '{"ultracode":true}'`, since the `--effort` flag rejects it.
|
||||
|
||||
Details:
|
||||
- New `effort` field on the create-session, quick-start, and Ralph-loop request schemas; threaded through `Session._effort` to both spawn paths (tmux `buildSpawnCommand` and direct-PTY `buildInteractiveArgs`), persisted in `SessionState.effort`, and restored on reboot recovery.
|
||||
- `buildEffortCliArgs()` is the single, allowlist-validated source for both carriers (injection-safe).
|
||||
- Settings UI adds an "Ultracode (multi-agent workflows)" option to the Thinking Effort dropdown; the frontend no longer emits `CLAUDE_CODE_EFFORT_LEVEL`.
|
||||
- Legacy migration: sessions persisted with the old env var are auto-migrated into the new `effort` field, and the stale tmux env var is unset so respawned panes are no longer locked.
|
||||
- Adds `test/effort-injection.test.ts` (13 cases) covering carrier mapping, injection guards, args building, and constructor migration.
|
||||
|
||||
## 0.8.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- Event-loop responsiveness fix, mobile image upload, response-viewer polish, and a mobile-UI trim.
|
||||
- **fix: avoid event-loop stalls from synchronous tmux/ps calls (#100):** The session manager ran `execSync` for tmux mouse-mode toggles, `list-panes`, and `ps`/`pgrep` resource-stat queries on the main thread. Under multi-session / many-pane load these blocking spawns froze Node's single event loop, stalling SSE broadcasts and PTY I/O (the ":3000 briefly unreachable, process never restarts" class of incident). Converted those calls to async `execAsync` and updated all callers to `await`. Added a lightweight `utils/event-loop-monitor.ts` that samples loop-delay and logs when a stall threshold is exceeded, started on web-server boot and stopped on shutdown — so future regressions leave a timestamped, quantified log line instead of vanishing silently.
|
||||
- **feat(web): mobile image upload to active session via paste dialog (#101):** The mobile keyboard-accessory paste dialog now attaches images, not just text — via a native picker (`accept=image/*` → camera / photo library / files) plus best-effort capture of images pasted into the textarea. Both paths reuse the existing `_uploadAndInsertImages()` → `POST /api/sessions/:id/paste-image` pipeline. Images are re-encoded client-side before upload (PNG→PNG to preserve transparency, everything else→JPEG, animated GIFs passed through untouched) so the bytes always match their declared extension — fixing the Android/MIUI case where a WebP/HEIF mislabeled as `image/jpeg` passed the extension allowlist but failed the server's magic-byte check. The server logs a precise diagnostic on any remaining magic-byte mismatch.
|
||||
- **feat(web): response-viewer transcript fallback + code-block rendering (#102):** A substantial response-viewer styling overhaul — proportional prose font (monospace kept for code), refined heading/code/blockquote/list styling, readable max content width, and a smoother slide-in animation; the `.rv-text` rules now also apply to `.response-viewer-body` so transcript-missing fallback content gets the same typography. Plus a `_renderMarkdown` null-safety fix (`text` → `src = text || ''`).
|
||||
- **feat(web): remove /compact button from the mobile keyboard accessory bar:** Dropped `/compact` from both the simple and extended accessory-bar layouts and the associated action handling. `/clear` retains its double-tap confirmation. Verified on a touch-emulated viewport that neither layout renders a compact action.
|
||||
|
||||
## 0.7.1
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- **fix(respawn): auto-accept now fires on plan approvals after `Worked for X` line, and on AskUserQuestion menus**
|
||||
|
||||
Two related blockers in the respawn controller's auto-accept path:
|
||||
- Modern Claude Code emits `✻ Worked for Xm Ys` immediately before a plan-approval menu. `_detectCompletionMessage()` cancelled the auto-accept timer and `canAutoAccept()` then rejected on `completionMessageTime !== null`, so plan approvals **never** auto-accepted — the 10 s completion-confirm timer instead started a respawn cycle while the menu sat unanswered.
|
||||
- The same logic in `signalElicitation()` set a hard flag that blocked auto-accept whenever Claude Code fired the `elicitation_dialog` hook, contradicting the in-UI hint ("Auto-accept presses Enter for plan approvals **and default question options**"). AskUserQuestion menus were therefore never auto-accepted either.
|
||||
|
||||
Fix:
|
||||
- `_detectCompletionMessage()` no longer cancels the auto-accept timer; the auto-accept pre-filter is now the authoritative "is there a numbered selection menu?" gate.
|
||||
- `canAutoAccept()` and the AI-plan-check callback both accept `'watching'` AND `'confirming_idle'` states (covers the single-PTY-burst case where `Worked for` and the menu arrive together — `_detectCompletionMessage` returns early before the substantial-output check can demote state back to watching). `sendAutoAcceptEnter()` self-transitions back to `'watching'` before sending Enter.
|
||||
- `signalElicitation()` is now an affirmative hint that primes the auto-accept timer instead of blocking. Still gated on `config.autoAcceptPrompts` AND state ∈ {`watching`, `confirming_idle`} — never fires Enter when respawn is off or auto-accept is disabled.
|
||||
- AI plan-check prompt broadened to recognize AskUserQuestion / elicitation menus as valid for auto-accept (the verdict name `PLAN_MODE` is preserved for compatibility but now means "auto-accept this selection menu").
|
||||
- Removed the now-unused `elicitationDetected` field and its assignments.
|
||||
|
||||
Two new regression tests cover both the separate-PTY-chunk and single-PTY-chunk cases; the previously misleading "should NOT send Enter when completion message was detected" test was renamed and re-scoped to clarify it tests the **no-menu** path (which still correctly rejects via the pre-filter).
|
||||
|
||||
**docs(web): correct `sendPendingCtrlL` comment** — removed the stale "called by foo/bar" note from the dead-call-graph helper after #99.
|
||||
|
||||
## 0.7.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- Response viewer & terminal-stability improvements, plus test/error-handling hardening.
|
||||
- **Copy button on code blocks (#98):** Every fenced code block in the response viewer now has a one-click copy button pinned to its top-right, outside the `<pre>` scroll container so it stays put during horizontal scroll. ASCII diagrams keep their line-wrap toggle alongside it. Copy prefers the async Clipboard API and falls back to a hidden-textarea + `execCommand` path, so it works over plain HTTP (tunnel) too, with a brief ✓/✕ feedback state.
|
||||
- **Fix: stop auto-sending Ctrl+L from session-selection paths (#99):** A fast page refresh or SSE reconnect could fire two programmatic Ctrl+L (`\x0c`) sends within Claude Code 2.x's "clear conversation" confirmation window, silently wiping the active conversation. Removed the automatic Ctrl+L sends from `selectSession()`, `restoreTerminalSize()`, and the dead `sendPendingCtrlL()` path; redraws now rely on resize/SIGWINCH. User-initiated Ctrl+L still works. Trade-off: an occasional transient stale Ink frame right after refresh that self-heals on the next keypress — far preferable to silent data loss.
|
||||
- **Test & error-handling hardening (#97):** Repaired route-test harness error rendering via a dedicated `route-error-handler.ts`, and stopped the AI idle/plan checkers from spawning real processes during tests.
|
||||
|
||||
## 0.6.12
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Fix new-session crash after a tmux upgrade and isolate Codeman sessions on a dedicated tmux socket.
|
||||
- **Pane file-descriptor limit**: raise `ulimit -Sn` before launching the CLI (in both the spawn and respawn paths) so the newer tmux + macOS launchd combination — which hands panes a low soft `nofile` limit (256) that recent Claude Code refuses to start under — no longer kills every freshly spawned session on startup.
|
||||
- **Single-socket isolation**: all Codeman-owned tmux sessions now live on a dedicated socket (`tmux -L codeman`, overridable via `CODEMAN_TMUX_SOCKET`), fully separated from the user's default tmux server. The socket name is validated and shell-escaped at every call site.
|
||||
- **Drop the drift-prone per-session `tmuxSocket` field**: session reconciliation collapses to a single `list-panes` query against the one socket, eliminating live sessions being wrongly marked dead ("session not found") and duplicate "Restored:" tabs. Stale per-session socket tags and duplicate records are cleaned from disk on load (dedup by `muxName`, keeping the real entry over `restored-` placeholders).
|
||||
- **Route remaining bare-`tmux` call sites through the socket**: the window-size query on re-attach (previously fell back to 120×40 and lost scrollback) and the send-key route (Shift+Enter / Ctrl+Enter newline).
|
||||
- **SSH chooser scripts** (`tmux-manager.sh`, `tmux-chooser.sh`) route every tmux call through the dedicated socket.
|
||||
|
||||
## 0.6.11
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Resume Conversation: fixes and folder drill-down.
|
||||
- **fix(history)**: `decodeProjectKey()` now uses longest-join-first backtracking with on-disk validation, so sibling directories sharing a prefix (e.g. `diary/` vs `diary-app/`) resolve to the correct path. Previously the greedy shortest-match decoder picked the shorter name and bailed, surfacing `$HOME` in the Resume Conversation list and resuming into the wrong folder. Greedy decode is kept as a fallback so history for deleted projects still resolves. (#92)
|
||||
- **fix(tabs)**: Drop the client-side resurrection of ended-session tabs. The old code cached open tabs in `localStorage` and rebuilt them as grayed-out stubs whenever the server no longer knew them, which left phantom tabs after closing a session on another device. The server is now the single source of truth; legacy `localStorage` keys are purged on init. Net -44 / +6 lines. (#93)
|
||||
- **feat(history)**: New "View all in this folder" drill-down on Resume Conversation. `GET /api/history/sessions` accepts `projectKey` (validated against `^[A-Za-z0-9_-]+$` before any filesystem access), `offset`, and `limit`; single-folder mode bypasses the 50-cap and returns `{ sessions, total }`. Frontend adds a modal listing 20 sessions per page with a "Show more" pagination button. Modal items omit their own "View all" button to prevent recursive entry points. (#94)
|
||||
|
||||
## 0.6.10
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- ## Security: paste-image endpoint hardening (#90)
|
||||
|
||||
Addresses seven findings from the dismissed review of #84. Most exposed in tunneled deployments where `CODEMAN_PASSWORD` is set but the server is reachable beyond localhost.
|
||||
- **CSRF protection** on `POST /api/sessions/:id/paste-image`. Requires `Origin`/`Referer` to match `req.host`; non-browser clients (no `Origin` and no `Referer`) must send `X-Codeman-CSRF`. Defeats cross-origin `<form enctype="multipart/form-data">` submits that would otherwise plant arbitrary bytes into the victim's `.claude-images/` while their session cookie is live.
|
||||
- **Magic-byte validation** on uploaded images. Sniffs the first 12 bytes against PNG/JPEG/GIF/WebP/BMP signatures and rejects 415 on mismatch. Polyglot HTML-or-SVG-with-image-MIME no longer round-trips through the endpoint.
|
||||
- **Symlink-safe writes** on `.claude-images/`. `lstat` before the write, non-recursive `mkdir`, `O_EXCL|O_NOFOLLOW` on file open. A `node_modules` postinstall (or the agent itself) planting `.claude-images -> ~/.ssh/` no longer redirects pastes outside `workingDir`.
|
||||
- **Multipart parser swap** to `@fastify/multipart` with `limits: { fileSize: 10MB, files: 1, fields: 4 }`. Replaces a hand-rolled boundary scanner that matched the literal boundary anywhere in the body, hard-coded `\r\n` (silently corrupting LF-only clients), and had no part-count cap.
|
||||
- **Rate limit + GC**: token-bucket (30/min per IP+session) and hourly GC of `paste-*` files older than 7 days from each live session's `.claude-images/`. New `paste-image-gc.ts` started/stopped from `WebServer.start/stop`.
|
||||
- **Collision-free filenames**: `paste-${Date.now()}-${randomBytes(4)}${ext}`. Two tabs pasting in the same millisecond no longer silently last-write-wins.
|
||||
- **Bracketed-paste preservation**: text-only paste in `image-input.js` now goes through `terminal.paste(text)` instead of `sendInput(text)`, so xterm preserves `CSI 200~ ... CSI 201~` markers — Claude Code uses them as part of its prompt-injection defenses.
|
||||
|
||||
## Fix: duplicate multipart parser conflict
|
||||
|
||||
Removed a duplicate multipart content-type parser left behind after the swap above. The duplicate registration conflicted with `@fastify/multipart`'s own parser; uploads now flow through the plugin exclusively.
|
||||
|
||||
## WebGL renderer auto-fallback hardening (#91)
|
||||
|
||||
Follow-ups on the longtask auto-fallback shipped in #83.
|
||||
- `PerformanceObserver` is now disconnected on `onContextLoss` as well as on the trip path. Previously the observer outlived its disposed addon after a context loss, holding a closure reference over every longtask the page emitted.
|
||||
- Thresholds (`200ms / 3 longtasks / 30s window / 5s grace / 7d sticky-disable`) are hoisted to `WEBGL_FALLBACK` in `constants.js`. No more inline literals.
|
||||
- New `evaluateWebGLLongTaskTrip()` pure helper splits the rolling-window arithmetic from the `PerformanceObserver` callback so the trip math is unit-testable. New `test/webgl-fallback.test.ts` (9 tests, port 3166): trip inside window, no-trip when spread, sub-threshold filtering, stale-entry pruning, cumulative counting across batches, observer-dispose idempotency.
|
||||
|
||||
## CI: server boot smoke test
|
||||
|
||||
GitHub Actions now boots the server as a final step after typecheck/lint/format. Catches production-only ESM/CJS regressions that `tsx` masks in dev.
|
||||
|
||||
## Docs
|
||||
|
||||
`CLAUDE.md` frontend-module table updated to include `image-input.js` (overlooked when #84 landed).
|
||||
|
||||
## 0.6.9
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Terminal renderer hardening, SSE bandwidth cut, image paste, and a security tightening on the new live filter:
|
||||
- **Multi-primitive yield for write pacing** (#85): replaces six raw `requestAnimationFrame` callsites in the xterm.js write pipeline with a yielding helper that races `requestAnimationFrame`, `setTimeout(50)`, and a tick Worker. Keeps the terminal responsive when the tab is backgrounded or occluded — Chrome's intensive-throttling no longer stalls long writes.
|
||||
- **WebGL longtask auto-fallback** (#83): a `PerformanceObserver` watches for ≥200ms WebGL frames; three within a 30s window disposes the WebGL addon and falls back to the canvas renderer. Decision is persisted in localStorage for 7 days, and `?webgl=force` clears it.
|
||||
- **Per-client live SSE subscription filter** (#86): each connected client gets a stable UUID and can narrow its terminal stream to one session via `POST /api/events/subscribe` — no EventSource reconnect on tab switches. Cuts SSE bandwidth roughly N× when N sessions are open. Lifecycle/metadata events (`session:*`, `case:*`, `ralph:*`, `hook:*`) now broadcast to every client so sidebars stay in sync.
|
||||
- **Image paste and drag-and-drop into the terminal** (#84): `Ctrl+V` and dropped images upload to `POST /api/sessions/:id/paste-image`, save under `${workingDir}/.claude-images/paste-${ts}.${ext}` and type the path into the terminal. Hard 10MB cap, server-generated filename (no traversal), `.svg` deliberately excluded from the allowlist to avoid a same-origin XSS path through `file-raw`.
|
||||
- **SSE clientId validation**: the per-client identifier introduced in #86 is now constrained to `[A-Za-z0-9_-]{8,64}` at both ingress points. Without this, an authenticated attacker could send another tab's clientId to silently evict it from broadcasts, mutate any clientId's session filter to blackhole the victim's terminal stream, or grow `sseClientsById` unboundedly via long IDs. The subscribe payload is also capped at 64 session entries of ≤128 chars each.
|
||||
|
||||
## 0.6.8
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Finish the hostname-aware notification plumbing started in 0.6.7 and lock down the recent UI/runtime fixes with regression tests.
|
||||
- Browser Notification API (OS-level desktop pop-ups, layer 3 of the 5-layer notification system) now uses `${originalTitle}: ${title}` instead of the hardcoded `Codeman:` literal — so multi-host users running Codeman on laptop / dev box / NAS see `codeman:<host>: <event>` consistently across tab title, tab-flash, Web Push, and OS notifications.
|
||||
- Inline session rename hardened against three corner cases: IME composition commits (Chinese pinyin Enter no longer ships half-composed text as the session name), mid-rename SSE deletion (orphaned `<input>` no longer 404s on blur), and double-fire on stuck settle-once flag (closure-local `settled` boolean replaces the boolean instance flag).
|
||||
- Test coverage backfilled for two prior shipped fixes:
|
||||
- `<title>codeman:<host></title>` server-side templating (#82): 8 tests covering default `os.hostname()`, `--title-hostname` override, HTML-escape against `<script>`-style breakout, ampersand non-double-encoding, and template-tail byte-identical invariance.
|
||||
- tmux size-query helper (#80): 15 tests covering the browser-resize-between-attaches happy path, the query-then-die race, zero/negative/empty/non-numeric output fallbacks, and argv-form/timeout assertions that lock down the no-shell-interpolation guarantee. Inline 14-line query block extracted into a named `queryTmuxWindowSize()` export in `session.ts` so the test surface is a pure function.
|
||||
- Regression coverage added for `stripInkRedrawBloat` route helper.
|
||||
- CLAUDE.md and README.md updated to document dual-CLI env-prefix discipline (`CLAUDE_CODE_*` vs `OPENCODE_*`), the `xterm-zerolag-input` published-package side-effect of overlay edits, and the unified hostname prefix across tab title / tab-flash / OS notifications.
|
||||
|
||||
## 0.6.7
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- - **fix(client): preserve inline rename input across tab re-renders** (#81) — Right-click → rename on a session tab no longer loses keystrokes when SSE traffic from sibling sessions triggers a tab re-render. Adds an `_inlineRenameActive` guard at the top of `renderSessionTabs()` and `_fullRenameSessionTabs()` so the in-progress input isn't destroyed mid-typing. Also fixes a latent double-fire of `finishRename` (blur + Enter could both invoke it). Drive-by: safer DOM child clearing in place of `innerHTML = ''`.
|
||||
- **feat: hostname-aware window title** (#82) — The browser tab title is now `codeman:<hostname>` instead of the bare `Codeman` literal, so users running Codeman on multiple hosts (laptop, dev box, NAS) can tell at a glance which tab points at which backend. New `--title-hostname <name>` CLI flag overrides the detected `os.hostname()` when it's noisy or you want a cosmetic name. The title is templated into the served HTML on first byte (with narrow HTML escaping), so it's correct from the first paint and works without JavaScript. Title-flash logic now respects the per-host title.
|
||||
- **perf: larger terminal tail on tab switch** — `TERMINAL_TAIL_SIZE` raised from 128KB to 1MB. When switching back to a busy session tab you now get ~8× more scrollback restored immediately.
|
||||
- **fix: preserve response text in Ink redraw stripping** — `stripInkRedrawBloat()` rewritten from a first-VPA approach to cluster-based detection. The previous algorithm assumed all VPA escapes after the first one belonged to a single redraw region and discarded everything in between, which silently lost 100KB+ of legitimate Claude response text once a render had occurred. The new approach groups VPAs into clusters separated by ≥8KB gaps and only collapses clusters spanning ≥32KB, so streamed response content between redraw bursts is preserved.
|
||||
- **docs**: `CLAUDE.md` Additional Commands gains the `--title-hostname` row; `README.md` gets a "Hostname-Aware Window Title" subsection under Multi-Session Dashboard.
|
||||
|
||||
## 0.6.6
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- **Terminal scrollback significantly increased** — both the xterm.js viewport and the tmux backing buffer were bottlenecking how far back you could scroll. Three changes:
|
||||
- `DEFAULT_SCROLLBACK` raised from 20000 → 50000 lines (xterm.js, main terminal). The previous bump from 5000 only helped users with empty localStorage; existing users were stuck on whatever value they first picked up. The loader now treats `DEFAULT_SCROLLBACK` as a floor — if your stored value is below the new minimum, you're raised to it automatically.
|
||||
- Subagent / teammate terminals (`panels-ui.js`) were stuck at 5000; now use the same `DEFAULT_SCROLLBACK` constant (50000).
|
||||
- New tmux sessions now run with `history-limit 50000` (tmux defaults to 2000). This matters for hard-reload / re-attach — without it, only the last ~2000 lines survive the round-trip back into a fresh xterm.
|
||||
|
||||
**Tmux flicker on session re-attach fixed (PR #80 by @aakhter)**: the PTY now queries the existing tmux window size via `tmux display -p` before spawning, instead of hardcoding 120x40. Previously, every re-attach forced tmux to resize down to 120x40, causing a visible flicker and one frame of scrollback loss. The `-x 120 -y 40` flag was also dropped from `tmux new-session` so the initial size matches the first attaching client. Uses `execFileSync` (not shell) for safety and falls back to 120x40 on any error.
|
||||
|
||||
**Docs**: CLAUDE.md now documents two recurring foot-guns — the `xterm-zerolag-input` overlay code is duplicated between `packages/xterm-zerolag-input/src/` and inline inside `src/web/public/app.js`, so any overlay change must touch both; and the COM workflow explicitly includes a post-push `gh run watch` step to confirm CI before considering the release done.
|
||||
|
||||
## 0.6.5
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- **Mobile fix**
|
||||
- Android virtual keyboard: space character was silently dropped on touch devices using GBoard / SwiftKey / similar IMEs. Root cause: the input-event handler in `terminal-ui.js` treated any whitespace-only textarea value as proof that xterm had already processed the input. A lone space (`' '.trim() === ''`) tripped this guard, so the space was consumed but never forwarded. Now skips only when the textarea is truly empty (or whitespace from a non-space key). Reported and diagnosed by @coolk8 in #79.
|
||||
|
||||
**Docs**
|
||||
- `CLAUDE.md`: added Zod `.optional()`-vs-`null` gotcha (recurring trap from 0.6.3 / 0.6.4 incidents) and a more visible warning against running bare `npm test` (kills the host tmux session).
|
||||
- `docs/local-echo-overlay-plan.md`: marked SHIPPED, corrected xterm version reference (v5.3.0 → `@xterm/xterm` ^6.0.0).
|
||||
|
||||
## 0.6.4
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Fix "Failed to enable respawn: Invalid request body" error when selecting infinity duration (∞) in the respawn modal. Frontend was sending `durationMinutes: null`, which Zod's `.optional()` schema rejected (it accepts `undefined` only). The body now omits the field when no duration is selected.
|
||||
|
||||
## 0.6.3
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- **Fix**
|
||||
- Allowlist `opusContext1mEnabled` in `SettingsUpdateSchema`. Without this entry, the strict schema rejected `PUT /api/settings {"opusContext1mEnabled":...}` with `INVALID_INPUT`, so the toggle's value never persisted across reloads. The frontend was already reading and writing this key (`settings-ui.js:336/1137`, `session-ui.js:340`), so saves were silently failing — users never noticed because the load path falls back to `false` on missing keys, hiding the bug. (#78)
|
||||
|
||||
## 0.6.2
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- **Mobile UX**
|
||||
- Resume Conversation list (welcome page) reworked for narrow screens: 2-line title clamp so more of the first prompt is visible; case-aware subtitle that renders `#caseName` (or `#caseName/sub`) when `workingDir` matches a known case, otherwise falls back to the directory basename; inline `⋯` toggle that expands a detail panel with full prompt, full path, timestamp, size, and short session id; `/Users/<user>/` now collapses to `~/` alongside `/home/<user>/`. (#77)
|
||||
- Response viewer: ASCII diagram wrap toggle, dedicated mobile code-block layout, and chrome-stripping fallback when the model wraps its reply in extra markup. (#75)
|
||||
- Mobile keyboard accessory bar no longer triggers vertical scroll. (#72)
|
||||
|
||||
**Sessions & settings**
|
||||
- New `thinkingEffort` setting on session creation, with `xhigh` option and `/effort max` mobile shortcut. (#73)
|
||||
- `thinkingEffort` is now allowlisted in `SettingsUpdateSchema` so it round-trips through PATCH /api/settings.
|
||||
- `envOverrides` (`CLAUDE_CODE_*` / `OPENCODE_*`) are now passed to Claude via tmux env exports at spawn time instead of being written to `<case>/.claude/settings.local.json`. Eliminates UI/disk drift; the value lives on `Session._envOverrides`, is exported by `tmux-manager.buildEnvExports()`, and is persisted in `SessionState.envOverrides`. (#74)
|
||||
|
||||
**Fixes**
|
||||
- Eye icon (active-session indicator) now follows `/clear` to the new Claude conversation instead of getting stuck on the previous transcript. (#76)
|
||||
- `tmux-manager.reconcileSessions` now uses `|` as the field separator, fixing parsing when session names contain other delimiters. (#71)
|
||||
|
||||
**Docs**
|
||||
- CLAUDE.md: added `npm run knip` to the dead-code sweep table and a `Common Gotchas` entry documenting the `envOverrides` → tmux export flow.
|
||||
|
||||
## 0.6.1
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Internal cleanup and release hygiene:
|
||||
- **Dead-code sweep via knip**: added `knip.json` for dead-code detection and ran a full sweep — removed unused test files, unused scripts, and narrowed internal module exports to the minimum surface area actually consumed.
|
||||
- **Lockfile drift prevention**: `version-packages` now runs `npm install --package-lock-only` and verifies the lockfile is in sync via `scripts/check-lockfile-sync.mjs`; CI runs the same check on every push/PR so version drift fails the build instead of reaching production. Resolves the `package-lock.json` / `package.json` version mismatch that shipped in 0.6.0.
|
||||
- **Docs tightening**: archived 22 completed plan docs from `docs/`, corrected file/handler counts in `CLAUDE.md`, documented the lockfile step in the COM workflow, and removed footer redundancy.
|
||||
|
||||
## 0.6.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- Community contributions from @aakhter:
|
||||
- **feat (#66): Tab reorder shortcuts** — `Ctrl+Shift+{` and `Ctrl+Shift+}` move the active session tab left/right, matching WezTerm convention. Order persists across reloads via `saveSessionOrder()`.
|
||||
- **feat (#67): Active tab visibility + Alt+N badges** — active tab now has a bright green border with color-matched glow, and the first 9 tabs display number badges hinting at the `Alt+N` switch shortcut. Badges update on reorder/rerender.
|
||||
- **feat (#68): Clipboard API** — new `POST /api/clipboard` accepting `{text}` broadcasts a `clipboard:write` SSE event; connected browsers attempt `navigator.clipboard.writeText()` with a manual-copy modal fallback when the page isn't focused. Auth-protected via the standard middleware. Useful for pushing snippets from remote sessions to the user's local clipboard.
|
||||
- **fix (#65): Android Shift+key double character** — pressing `Shift+A` on attached Android keyboards no longer produces "AA". Tracks xterm-handled keydown timestamps and skips the orphaned-input listener for 50ms after a real keydown, while still catching Gboard symbol-keyboard inputs (keyCode 229).
|
||||
|
||||
## 0.5.13
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Fix "Case path not found" error in Quick Start when `~/codeman-cases/` does not exist (issue #64). Two bugs in `session-ui.js`:
|
||||
- `runClaude()` auto-create read `createCaseData.case`, but `POST /api/cases` returns `{ success, data: { case } }` — corrected to `createCaseData.data.case`.
|
||||
- `runShell()` had no auto-create logic and would immediately throw on a missing case directory — now mirrors `runClaude()`'s create-on-demand flow.
|
||||
|
||||
## 0.5.12
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Fix quick-start to resolve linked cases before codeman-cases fallback. `/api/quick-start` was always resolving `caseName` against `CASES_DIR`, ignoring entries in `~/.codeman/linked-cases.json`. Sessions started via quick-start now correctly honour linked external project directories, consistent with regular case routes.
|
||||
|
||||
## 0.5.11
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Community contributions and security hardening:
|
||||
- Mobile response viewer: native-scroll panel for reading full Claude responses with markdown rendering via marked.js (PR #62)
|
||||
- PWA support: service worker caching, web app manifest, and Android home screen install (PR #59)
|
||||
- Named Cloudflare tunnel support (PR #58)
|
||||
- Markdown rendering for response viewer with HTML sanitization (XSS prevention) — strips dangerous elements, event handlers, and javascript: URIs
|
||||
- Service worker switched from stale-while-revalidate to network-first caching so deploys take effect immediately
|
||||
- Content-Disposition filename sanitization to prevent header injection in file downloads
|
||||
- Expose session.muxName public getter, replace unsafe `as any` cast in session-routes
|
||||
- Static import for execFile in session-routes
|
||||
- Keyboard shortcut updates: Alt+1-9 tab switching, Shift+Enter newline
|
||||
- Repo restructure for cleaner GitHub landing page
|
||||
- Mobile logo, expandable history, session resume fixes
|
||||
|
||||
## 0.5.10
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- fix: allow bracket characters in model validation regex so models like opus[1m] (1M context window) are accepted instead of silently dropped. Quote the model flag value in tmux spawn commands to prevent bash glob expansion of bracket patterns.
|
||||
|
||||
docs: update macOS launchd instructions to use `launchctl bootstrap` instead of deprecated `load`. Clean up README install and service sections.
|
||||
|
||||
## 0.5.9
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Mobile keyboard accessory bar: add configurable "Extended Keyboard Bar" setting (Settings > Display > Input) that toggles between simple mode (up/down arrows, /init, /clear, /compact, paste, dismiss) and extended mode (adds left/right arrows, Tab, Shift+Tab, Ctrl+O, Alt+Enter, Esc). Default is simple mode. Setting is device-specific (not synced to server).
|
||||
|
||||
Restyle dismiss button: muted steel-blue tone, fills remaining bar space via flex, larger tap target. Arrow buttons now blue.
|
||||
|
||||
Fix paste overlay visibility on mobile: dialog repositioned to top of screen (15vh from top) so the virtual keyboard doesn't cover it. Textarea enlarged for better usability.
|
||||
|
||||
(Also includes all v0.5.8 changes: case reorder/delete, XSS sanitization, auto-attach PTY on restart, mobile keyboard buttons, macOS installer fixes, terminal flicker fix, state store collision fix.)
|
||||
|
||||
## 0.5.8
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Case management: add Manage tab with reorder (up/down arrows) and delete for cases; linked cases are unlinked (folder preserved), CASES_DIR cases are permanently deleted. New endpoints: DELETE /api/cases/:name, PUT /api/cases/order. SSE events: case:deleted, case:order-changed.
|
||||
|
||||
Security: sanitize case names from filesystem with /^[a-zA-Z0-9_-]+$/ regex before returning from GET /api/cases to prevent XSS via maliciously-named directories reaching frontend inline onclick handlers.
|
||||
|
||||
Auto-attach PTY: server now calls startInteractive() for recovered tmux sessions during startup so all sessions resume capturing output immediately after deploy, instead of waiting for client selection. Frontend auto-attach condition relaxed from (pid===null && status==='idle') to (pid===null && !\_ended).
|
||||
|
||||
Mobile keyboard accessory: add Shift+Tab, Tab, Esc, Alt+Enter, Left/Right arrow, and Ctrl+O buttons.
|
||||
|
||||
Terminal: fix flicker regression by moving viewport clear inside dimension guard.
|
||||
|
||||
State store: fix temp file collisions on concurrent writes.
|
||||
|
||||
macOS: fix installer failures when piped via curl | bash, add HTML cache support, launchd service template, and trust dialog handling.
|
||||
|
||||
Housekeeping: remove accidentally committed dist/state-store.js build artifact.
|
||||
|
||||
## 0.5.7
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- feat: support "Default (CLI default)" option for model selection. Adds a new empty-value option to the model dropdown that defers to the CLI's own default model instead of forcing a specific model. Ensures empty defaultModel values are treated as undefined when passed to session creation and Ralph loop start, preventing empty strings from being sent as model flags.
|
||||
|
||||
## 0.5.6
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- fix: default new sessions to opus[1m] (1M context window) instead of plain opus (200k context)
|
||||
|
||||
## 0.5.5
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Add 1M Opus context quick setting — per-case and global toggle that writes `model: "opus[1m]"` to `.claude/settings.local.json` when creating new sessions. Fix mobile layout: banners (respawn, timer, orchestrator) between header and main content now visible by switching from margin-top on `.main` to padding-top on `.app`. Add tablet-optimized respawn banner styles and mobile phone banner refinements.
|
||||
|
||||
## 0.5.4
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Fix terminal flicker regression — re-add server-side DEC 2026 synchronized output wrapping around batched terminal data. Ink spinner frames (cursor-up + redraw cycles) do not emit their own DEC 2026 markers, so without the server wrapper each partial cursor update rendered individually causing visible flicker. Also: extract SSE stream management, session listener wiring, and respawn event wiring from server.ts into dedicated modules; deduplicate error message extraction across 7 files with shared getErrorMessage() helper; update SSE event count in CLAUDE.md (106 → 117).
|
||||
|
||||
## 0.5.3
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Readability refactor across 12 core files, extracting ~35 helper methods to reduce duplication:
|
||||
- state-store: extract serializeState(), split assembleStateJson() into focused sub-methods
|
||||
- session: extract \_resetBuffers() (3x dedup), \_clearAllTimers() (10 timer cleanups), \_handleJsonMessage()
|
||||
- ralph-tracker: extract completeAllTodos() (4x dedup), emitValidationWarning(), named similarity constants
|
||||
- subagent-watcher: extract markSubagentAsCompleted(), extractFirstTextContent(), emitToolResult(), findOldestInactiveAgent()
|
||||
- respawn-controller: extract recoveryResetToWatching(), canAutoAccept(), formatRemainingSeconds(), validatePositiveTimeout()
|
||||
- tmux-manager: replace 15 path.includes() with UNSAFE_PATH_CHARS regex, extract buildEnvExports/buildPathExport/\_configureOpenCode helpers
|
||||
- session-auto-ops: extract executeWhenIdle() shared retry helper, convert to options object, add validateThreshold()
|
||||
- app.js: add \_clearTimer() (11 call sites), \_isStaleSelect(), keyboard shortcut lookup table, \_cleanupPreviousSession(), \_resetAllAppState()
|
||||
- route-helpers: add readJsonConfig() (5 inline patterns replaced), validateSessionFilePath() (2 duplicated blocks replaced)
|
||||
|
||||
## 0.5.2
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Make buffer size limits configurable via CODEMAN\_\* environment variables (MAX_TERMINAL_BUFFER, TRIM_TERMINAL_TO, MAX_TEXT_OUTPUT, TRIM_TEXT_TO, MAX_MESSAGES), falling back to existing defaults. Allows users with fewer sessions or more RAM to tune buffer sizes without patching source.
|
||||
|
||||
Fix duplicate terminal output on tab switch to busy sessions by clearing the terminal before writing the new buffer.
|
||||
|
||||
Fix stale Ink CUP frames after tab switch by sending Ctrl+L to force a clean redraw.
|
||||
|
||||
Fix mobile CJK input handling: resolve textarea positioning, terminal flicker during composition, and layout overflow on small screens. Improve CJK composition lifecycle with better event handling and fallback flush timers.
|
||||
|
||||
## 0.5.1
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- refactor: codebase cleanup — extract route helpers, eliminate boilerplate, optimize hot paths
|
||||
- Add `parseBody()` helper to route-helpers.ts: validates request body against Zod schema with structured 400 error on failure, replacing 37 identical safeParse + error-check blocks across 10 route files
|
||||
- Add `persistAndBroadcastSession()` helper: combines persist + SessionUpdated broadcast into one call, replacing 5 repeated 2-line pairs
|
||||
- Migrate session-routes.ts to use `findSessionOrFail()` consistently (17 inline session lookups replaced) and `parseBody()` (12 patterns)
|
||||
- Migrate ralph-routes.ts to use `findSessionOrFail()` (9 lookups) and `parseBody()` (4 patterns)
|
||||
- Migrate 8 remaining route files to use `parseBody()` (21 patterns total)
|
||||
- Fix O(n log n) eviction in bash-tool-parser.ts: replace `Array.from().sort()[0]` with O(n) min-scan for oldest active tool
|
||||
- Extract `_debouncedCall()` utility in frontend: replaces 4 manual debounce patterns (7 lines each → 1 line) in app.js, panels-ui.js, ralph-panel.js
|
||||
- Net reduction: 208 lines removed across 16 files
|
||||
|
||||
## 0.5.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- Visual redesign with glass morphism, refined colors, and polished UI. Optimize history endpoint with buffer reuse and line iterator. Fix Ink frame search window (4KB→64KB) to prevent partial frames. Fix stale terminal data on tab switch via chunkedTerminalWrite cancellation. Improve history prompt extraction with expanded command filtering and tail scan fallback. Align case select group height to match dropdown. Fix no-control-regex lint error for ANSI strip pattern. Add browser-testing-guide to CLAUDE.md references.
|
||||
|
||||
## 0.4.7
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- feat: improve session navigability in history and monitor panel (closes #45)
|
||||
- History items now show the first user prompt as the title with the project path as a subtitle, making it much easier to distinguish sessions from the same project
|
||||
- The `/api/history/sessions` endpoint extracts the first user message from each transcript JSONL, stripping system-injected XML tags and command artifacts, truncating to 120 chars
|
||||
- Monitor panel session rows are now clickable — clicking navigates directly to that session's tab via `selectSession()`; Kill button retains independent behavior via `stopPropagation()`
|
||||
- Updated CLAUDE.md architecture tables to reflect Orchestrator Loop additions (14 route modules, 15 type files, orchestrator domain files, orchestrator-panel.js frontend module)
|
||||
- fix: stop subagent monitor windows from auto-opening on discovery
|
||||
- feat: add Orchestrator Loop with phased plan execution, live progress during plan generation, and toolbar button (hidden until fully tested)
|
||||
- fix: patch 3 production bugs found during deep audit
|
||||
- fix: restore mobile terminal scrollback using JS scrollLines() instead of broken native scroll
|
||||
|
||||
## 0.4.6
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Fix mobile keyboard scroll and layout issues:
|
||||
- Prevent iOS Safari from scrolling the page when typing with the keyboard open (position:fixed on .app + window.scroll reset)
|
||||
- Eliminate dead space between terminal and keyboard accessory bar by removing redundant CSS padding, tightening JS padding constant, and adding row quantization gap compensation
|
||||
- Fix toolbar overlapping terminal content when keyboard is hidden by adding proper padding-bottom to .main, including iOS Safari bottom bar offset
|
||||
- Strip Ink spinner bloat from terminal buffer before tailing
|
||||
- Fix resolveCasePath priority order and suppress JSON parse warnings
|
||||
|
||||
## 0.4.5
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Fix mobile keyboard toolbar positioning on iOS Safari: toolbar (Run/Stop/Run Shell) was hidden behind the accessory bar when virtual keyboard was active due to overlapping CSS positions. Remove the aggressive safety check in `updateLayoutForKeyboard()` that incorrectly dismissed keyboard state when iOS scrolled the visual viewport during typing. Add Safari-bar CSS offset to accessory bar so it properly stacks above the toolbar. Remove the double-counted Safari-bar offset when keyboard is visible since the JS transform already covers the full distance.
|
||||
|
||||
## 0.4.4
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- fix: mobile keyboard hides terminal content on iPhone
|
||||
|
||||
Fixed a bug where opening the virtual keyboard on iPhone left zero visible terminal space. Two independent mechanisms were both accounting for the keyboard height: `MobileDetection.updateAppHeight()` shrunk `--app-height` to the visual viewport height, while `KeyboardHandler.updateLayoutForKeyboard()` added a large `paddingBottom`. These double-counted, leaving negative space for the terminal (user saw accessory bar + toolbar but no terminal content).
|
||||
|
||||
Fix: `updateAppHeight()` now skips when the keyboard is visible, and `handleViewportResize()` restores `--app-height` to the pre-keyboard value on first detection (since MobileDetection's listener fires before KeyboardHandler's). On keyboard close, `--app-height` is re-synced to the current visual viewport.
|
||||
|
||||
## 0.4.3
|
||||
|
||||
### Patch Changes
|
||||
|
||||
@@ -6,11 +6,12 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
|
||||
|
||||
| Task | Command |
|
||||
|------|---------|
|
||||
| Dev server | `npx tsx src/index.ts web` |
|
||||
| Dev server | `npm run dev` (or `npx tsx src/index.ts web`) |
|
||||
| Type check | `tsc --noEmit` |
|
||||
| Lint | `npm run lint` (fix: `npm run lint:fix`) |
|
||||
| Format | `npm run format` (check: `npm run format:check`) |
|
||||
| Single test | `npx vitest run test/<file>.test.ts` |
|
||||
| Single test | `npm test -- test/<file>.test.ts` (or `npx vitest run --config config/vitest.config.ts test/<file>.test.ts`) — ⚠ **never** run bare `npm test`, see Testing section |
|
||||
| Build | `npm run build` (esbuild via `scripts/build.mjs`, NOT tsc — `tsc --noEmit` is type-check only) |
|
||||
| Production | `npm run build && systemctl --user restart codeman-web` |
|
||||
|
||||
## CRITICAL: Session Safety
|
||||
@@ -29,7 +30,7 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
|
||||
2. **Frontend changes**: Use Playwright to load the page and assert the UI renders correctly. Use `waitUntil: 'domcontentloaded'` (not `networkidle` — SSE keeps the connection open). Wait 3-4s for polling/async data to populate, then check element visibility, text content, and CSS values
|
||||
3. **Only after verification passes**, proceed with COM
|
||||
|
||||
The production server caches static files for 1 year (`maxAge: '1y'` in `server.ts`). After deploying frontend changes, users may need a hard refresh (Ctrl+Shift+R) to see updates.
|
||||
The production server caches static files for 1 year, `immutable` (`maxAge: '1y'` in `server.ts`). To avoid stale frontend after a deploy, `renderIndexHtml` runs `cacheBustAssets(html)` — it appends `?v=<mtime>` to **every same-origin `.js`/`.css`** reference (mtime memoized ~1s so a burst of renders is cheap; external/already-versioned/missing refs untouched). Because `index.html` is served `no-cache`, a **normal reload now picks up edited modules/styles — no hard refresh needed** (the gesture bundle is injected separately with its own `?v=`). If you add an asset referenced by an *absolute* URL or from JS rather than a `<script>/<link>` tag, it won't be auto-busted.
|
||||
|
||||
## COM Shorthand (Deployment)
|
||||
|
||||
@@ -48,11 +49,14 @@ When user says "COM":
|
||||
CHANGESET
|
||||
```
|
||||
Replace `patch` with `minor` or `major` as needed. Include `"xterm-zerolag-input": patch` on a separate line if that package changed too.
|
||||
3. **Consume the changeset**: `npm run version-packages` (bumps versions in `package.json` files and updates `CHANGELOG.md`)
|
||||
3. **Consume the changeset**: `npm run version-packages` (auto-bumps `package.json` files, updates `CHANGELOG.md`, runs `npm install --package-lock-only`, and verifies lockfile sync via `scripts/check-lockfile-sync.mjs` — all in one command; never hand-edit `CHANGELOG.md` or `package-lock.json` versions)
|
||||
4. **Sync CLAUDE.md version**: Update the `**Version**` line below to match the new version from `package.json`
|
||||
5. **Commit and deploy**: `git add -A && git commit -m "chore: version packages" && git push && npm run build && systemctl --user restart codeman-web`
|
||||
6. **Wait for CI**: after `git push`, find the run with `gh run list -L 1 --json databaseId,headBranch -q '.[0].databaseId'` and watch it with `gh run watch <id> --exit-status`. Confirm all checks pass before considering the release done.
|
||||
|
||||
**Version**: 0.4.3 (must match `package.json`)
|
||||
CI runs `npm run check:lockfile` on every push/PR, so lockfile drift fails the build even if the `version-packages` script is bypassed.
|
||||
|
||||
**Version**: 0.9.1 (must match `package.json`)
|
||||
|
||||
## Project Overview
|
||||
|
||||
@@ -68,26 +72,37 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
|
||||
## Additional Commands
|
||||
|
||||
`npm run dev` = dev server. Default port: `3000`. Commands not in Quick Reference:
|
||||
`npm run dev` = dev server. Default port: `3000` (override with `--port` or the `CODEMAN_PORT` env var). To run this beta isolated alongside a prod Codeman, use `scripts/run-beta.sh` (sets `CODEMAN_INSTANCE=beta` + `CODEMAN_PORT=5000`). Commands not in Quick Reference:
|
||||
|
||||
| Task | Command |
|
||||
|------|---------|
|
||||
| Dev with TLS | `npx tsx src/index.ts web --https` |
|
||||
| Override window title hostname | `npx tsx src/index.ts web --title-hostname <name>` (default: `os.hostname()` — `codeman:<name>` is used for tab title, title-flash, and OS desktop notification prefix) |
|
||||
| Bind a non-loopback host | `npx tsx src/index.ts web --host 0.0.0.0` (or `-H`; env `CODEMAN_HOST`; default `127.0.0.1`). Without `CODEMAN_PASSWORD` it **starts but warns loudly** — see Common Gotchas + `docs/security-architecture.md` |
|
||||
| Continuous typecheck | `tsc --noEmit --watch` |
|
||||
| Test coverage | `npm run test:coverage` |
|
||||
| Dead-code sweep | `npm run knip` (config in `knip.json`) |
|
||||
| Check public-asset formatting | `npm run check:public-assets` (prettier-checks `src/web/public/**` text assets; `scripts/check-public-assets.mjs`) |
|
||||
| Production start | `npm run start` |
|
||||
| Production logs | `journalctl --user -u codeman-web -f` |
|
||||
|
||||
**CI**: `.github/workflows/ci.yml` runs `typecheck`, `lint`, `format:check` on push to master (Node 22). Tests excluded (they spawn tmux).
|
||||
**CI**: `.github/workflows/ci.yml` runs `check:lockfile`, `typecheck`, `lint`, `format:check`, then a **server boot smoke test** (`tsx src/index.ts web --port 3151` must answer `/api/status` within 30s) on push to master/main and on PRs (Node 22). The unit test suite is excluded (it spawns tmux).
|
||||
|
||||
**Code style**: Prettier (`singleQuote: true`, `printWidth: 120`, `trailingComma: "es5"`). ESLint flat config (`eslint.config.js`) allows `no-console`, warns on `@typescript-eslint/no-explicit-any`. Ignores: `app.js`, `scripts/**/*.mjs`, `src/web/public/vendor/**`, `tools/**`, `remotion/**`.
|
||||
**Code style**: Prettier (`singleQuote: true`, `printWidth: 120`, `trailingComma: "es5"`). ESLint flat config (`config/eslint.config.js`) allows `no-console`, warns on `@typescript-eslint/no-explicit-any`. Ignores: `app.js`, `scripts/**/*.mjs`, `src/web/public/vendor/**`, `scripts/remotion/**`.
|
||||
|
||||
## Common Gotchas
|
||||
|
||||
- **Single-line prompts only** — `writeViaMux()` sends text+Enter separately; multi-line breaks Ink
|
||||
- **ESM only** — Never `require()`, use `await import()`. `tsx` masks CJS/ESM issues in dev but production breaks
|
||||
- **Package ≠ product name** — npm: `aicodeman`, product: **Codeman**. Release renames tags accordingly
|
||||
- **Global regex `lastIndex`** — Use `createAnsiPatternFull/Simple()` factories, not shared `g`-flag patterns in loops
|
||||
- **Global regex `lastIndex`** — Shared `g`-flag patterns in loops must reset `lastIndex = 0` first, or use the `execPattern()` helper in `utils/regex-patterns.ts` (resets automatically)
|
||||
- **`envOverrides` flow `CLAUDE_CODE_*` / `OPENCODE_*` env vars** — Set via `POST /api/sessions { envOverrides }`, stored on `Session._envOverrides`, exported by `tmux-manager.buildEnvExports()` at spawn time, persisted in `SessionState.envOverrides`. **Do NOT** write these to `<case>/.claude/settings.local.json` — that's the old path and creates UI/disk drift
|
||||
- **Effort is NOT an env var** — never carry effort as `CLAUDE_CODE_EFFORT_LEVEL`: the env var hard-locks effort and blocks in-session `/effort` switching (incl. ultracode). It flows as the dedicated `effort` payload field → `Session._effort` → `claude --effort <level>` for regular levels incl. `max` (the settings `effortLevel` key is `enum(["low","medium","high","xhigh"]).catch(undefined)` — `max` gets SILENTLY dropped there), or `claude --settings '{"ultracode":true}'` for ultracode (rejected by `--effort`). Both are soft defaults the user can override anytime. Legacy env-var entries are auto-migrated by the Session constructor and unset from tmux sessions in `applyEnvOverrides()`. See `buildEffortCliArgs()` in `session-cli-builder.ts`, tests in `test/effort-injection.test.ts`
|
||||
- **Dual-CLI prefix discipline** — Codeman supports both Claude Code and OpenCode (`claude-cli-resolver.ts` / `opencode-cli-resolver.ts`); env-var prefix is CLI-specific (`CLAUDE_CODE_*` vs `OPENCODE_*`) and the allowlist in `schemas.ts` enforces this. When adding settings, decide which CLI(s) it applies to and gate the env export accordingly — don't blindly forward both prefixes. See `docs/opencode-integration.md` for the OpenCode resolver design
|
||||
- **Zod `.optional()` rejects `null`** — accepts `undefined` only. When the frontend builds a request body with `JSON.stringify`, an explicit `null` field is preserved on the wire and fails validation with `INVALID_INPUT`. Convert `null` → `undefined` before stringifying (e.g. `field: value ?? undefined`), or declare the schema `.nullish()`. Real bugs caused: 0.6.4 (`durationMinutes` for ∞ respawn), and the same shape pattern hit `opusContext1mEnabled` in 0.6.3
|
||||
- **`xterm-zerolag-input` is duplicated** — the local-echo overlay lives in BOTH `packages/xterm-zerolag-input/src/` (published to npm as a standalone library for external consumers — see README "Published Packages") AND inline inside `src/web/public/app.js` (runtime copy the web UI actually loads, since the page ships as plain JS without a bundler). Any change to overlay behavior MUST be applied to both, or dev and prod diverge — and a public API break in the package warrants a separate version bump for `xterm-zerolag-input` in the changeset. Always test on mobile after touching it. See `docs/local-echo-overlay-plan.md`.
|
||||
- **Default bind is loopback-only; non-loopback without a password starts but warns** — since COD-29 (PR #107) the web server defaults to `--host 127.0.0.1` (was `0.0.0.0`). As of **0.9.0** binding a non-loopback host (`--host`/`-H`/`CODEMAN_HOST`) without `CODEMAN_PASSWORD` **no longer refuses to start — it starts and prints a loud warning** listing the fixes (set `CODEMAN_PASSWORD`, bind loopback + tunnel/`tailscale serve`, or `--allow-unauthenticated-network` / `CODEMAN_ALLOW_UNAUTHENTICATED_NETWORK=1` to acknowledge → terser note). Host classification is `isLoopbackBindHost()` in `network-auth-policy.ts`; the warn-vs-start logic is in `server.ts` `start()`; flags wired in `cli.ts`. ⚠️ Operational note: the production systemd unit runs `node dist/index.js web --https` with no `--host`, so it binds **localhost only** — reach it remotely via `tailscale serve`/tunnel to `127.0.0.1`, or add `Environment=CODEMAN_HOST=0.0.0.0` + `Environment=CODEMAN_PASSWORD=…` to `~/.config/systemd/user/codeman-web.service`. A loopback bind is reachable through a same-host tunnel (cloudflared/tailscale → `127.0.0.1`) but NOT by a browser hitting the box's LAN IP. Auth user defaults to `admin`. **Full model: `docs/security-architecture.md`.**
|
||||
- **Instance isolation / multi-instance attach danger** — data dir (`~/.codeman`) and tmux socket (`tmux -L codeman`) are PROCESS-WIDE and shared by every Codeman on the machine, derived from `CODEMAN_INSTANCE` via `src/config/instance.ts` (`getDataDir()`/`dataPath()`/`DEFAULT_TMUX_SOCKET`). ⚠️ A 2nd instance on the SAME socket **discovers and attaches PTYs to the first instance's live sessions** (`tmux -L codeman attach-session …`), resizing/mutating them — `$HOME` isolation is NOT enough (tmux is system-global). To run two instances, give each a distinct `CODEMAN_INSTANCE` (scopes BOTH dir+socket: `~/.codeman-<name>` + `-L codeman-<name>`), or set `CODEMAN_TMUX_SOCKET` + `CODEMAN_DATA_DIR` individually. **`CODEMAN_INSTANCE` defaults to empty = the production layout (`~/.codeman`, `-L codeman`, port 3000)**, so this branch is safe to ship to master without disturbing existing installs. To run THIS beta alongside prod, launch with `scripts/run-beta.sh` (`CODEMAN_INSTANCE=beta` + `CODEMAN_PORT=5000`) — it never collides with prod's data dir/socket/port. Any new `~/.codeman/...` path MUST go through `dataPath()`, never `join(homedir(), '.codeman', …)`.
|
||||
|
||||
**Import conventions**: Utils from `./utils`, types from `./types` (barrel), config from specific `./config/*` files.
|
||||
|
||||
@@ -99,26 +114,27 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
|--------|-----------|-------|
|
||||
| **Entry** | `src/index.ts`, `src/cli.ts` | |
|
||||
| **Session** | `src/session.ts` ★, `src/session-manager.ts`, `src/session-auto-ops.ts`, `src/session-cli-builder.ts`, `src/session-lifecycle-log.ts`, `src/session-task-cache.ts` | |
|
||||
| **Mux** | `src/mux-interface.ts`, `src/mux-factory.ts`, `src/tmux-manager.ts` | |
|
||||
| **Mux** | `src/mux-interface.ts`, `src/mux-factory.ts`, `src/tmux-manager.ts` ★ | |
|
||||
| **Respawn** | `src/respawn-controller.ts` ★ + 4 helpers (`-adaptive-timing`, `-health`, `-metrics`, `-patterns`) | Read `docs/respawn-state-machine.md` first |
|
||||
| **Ralph** | `src/ralph-tracker.ts` ★, `src/ralph-loop.ts` + 5 helpers (`-config`, `-fix-plan-watcher`, `-plan-tracker`, `-stall-detector`, `-status-parser`) | Read `docs/ralph-wiggum-guide.md` first |
|
||||
| **Orchestrator** | `src/orchestrator-loop.ts`, `src/orchestrator-planner.ts`, `src/orchestrator-verifier.ts` | Read `docs/orchestrator-loop-architecture.md` first |
|
||||
| **Agents** | `src/subagent-watcher.ts` ★, `src/team-watcher.ts`, `src/bash-tool-parser.ts`, `src/transcript-watcher.ts` | |
|
||||
| **AI** | `src/ai-checker-base.ts`, `src/ai-idle-checker.ts`, `src/ai-plan-checker.ts` | |
|
||||
| **Tasks** | `src/task.ts`, `src/task-queue.ts`, `src/task-tracker.ts` | |
|
||||
| **State** | `src/state-store.ts`, `src/run-summary.ts`, `src/session-lifecycle-log.ts` | |
|
||||
| **Infra** | `src/hooks-config.ts`, `src/push-store.ts`, `src/tunnel-manager.ts`, `src/image-watcher.ts`, `src/file-stream-manager.ts` | |
|
||||
| **Plan** | `src/plan-orchestrator.ts`, `src/prompts/*.ts`, `src/templates/claude-md.ts` | |
|
||||
| **Web** | `src/web/server.ts`, `src/web/sse-events.ts`, `src/web/routes/*.ts` (13 route modules incl. `ws-routes.ts` + barrel), `src/web/ports/*.ts`, `src/web/middleware/auth.ts`, `src/web/schemas.ts` | |
|
||||
| **Frontend** | `src/web/public/app.js` (~2.6K lines, core) + 5 infra modules (`constants.js`, `mobile-handlers.js`, `voice-input.js`, `notification-manager.js`, `keyboard-accessory.js`) + 6 domain modules (`terminal-ui.js`, `respawn-ui.js`, `ralph-panel.js`, `settings-ui.js`, `panels-ui.js`, `session-ui.js`) + 4 feature modules (`ralph-wizard.js`, `api-client.js`, `subagent-windows.js`, `input-cjk.js`) + `sw.js` | |
|
||||
| **Types** | `src/types/index.ts` → 13 domain files | See `@fileoverview` in index.ts |
|
||||
| **Web** | `src/web/server.ts`, `src/web/sse-events.ts`, `src/web/routes/*.ts` (15 route modules + barrel), `src/web/route-helpers.ts`, `src/web/ports/*.ts`, `src/web/middleware/auth.ts`, `src/web/schemas.ts` | |
|
||||
| **Frontend** | `src/web/public/app.js` (~3.4K lines, core) + 5 infra modules (`constants.js`, `mobile-handlers.js`, `voice-input.js`, `notification-manager.js`, `keyboard-accessory.js`) + 7 domain modules (`terminal-ui.js`, `respawn-ui.js`, `ralph-panel.js`, `orchestrator-panel.js`, `settings-ui.js`, `panels-ui.js`, `session-ui.js`) + 5 feature modules (`ralph-wizard.js`, `api-client.js`, `subagent-windows.js`, `input-cjk.js`, `image-input.js`) + `sw.js` | |
|
||||
| **Types** | `src/types/index.ts` (barrel) → 14 domain files; also `src/types.ts` root re-export | See `@fileoverview` in index.ts |
|
||||
|
||||
★ = Large file (>50KB). All files have `@fileoverview` JSDoc — read that before diving in.
|
||||
★ = Large file (>50KB). All files have `@fileoverview` JSDoc — read that before diving in. Discovery aid: `grep -l '@fileoverview' src/web/routes/*.ts` lists all route modules; same grep works for `src/types/`, `src/web/public/*.js`.
|
||||
|
||||
**Local package**: `packages/xterm-zerolag-input/` — local echo overlay for xterm.js; copy embedded in `app.js`.
|
||||
|
||||
**Config**: `src/config/` — 9 files. Import from specific files, not barrel.
|
||||
**Config**: `src/config/` — 10 files. Import from specific files, not barrel.
|
||||
|
||||
**Utilities**: `src/utils/` — re-exported via index. Key: `CleanupManager`, `LRUMap`, `StaleExpirationMap`, `BufferAccumulator`, `stripAnsi`, `Debouncer`, `KeyedDebouncer`. Also: `claude-cli-resolver`/`opencode-cli-resolver` (CLI path resolution), `string-similarity` (fuzzy matching), `regex-patterns` (ANSI/token/spinner patterns), `assertNever` (exhaustive checks).
|
||||
**Utilities**: `src/utils/` — re-exported via index. Key: `CleanupManager`, `LRUMap`, `StaleExpirationMap`, `BufferAccumulator`, `stripAnsi`, `Debouncer`, `KeyedDebouncer`. Also: `claude-cli-resolver`/`opencode-cli-resolver` (CLI path resolution), `string-similarity` (fuzzy matching), `regex-patterns` (ANSI/token/spinner patterns), `assertNever` (exhaustive checks), `token-validation` (auth tokens), `nice-wrapper` (process priority).
|
||||
|
||||
### Data Flow
|
||||
|
||||
@@ -133,9 +149,11 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
|
||||
**Idle detection**: Multi-layer (completion message → AI check → output silence → token stability). See `docs/respawn-state-machine.md`.
|
||||
|
||||
**Orchestrator**: State machine that turns a user goal into a phased plan and drives it to completion: `idle → planning → approval → executing → verifying → (replanning) → completed/failed`. `OrchestratorLoop` (engine) delegates plan generation to `orchestrator-planner` and per-phase verification gates to `orchestrator-verifier`, executing phases via team agents/`task-queue`. State persists under the `orchestrator` key in `state.json`. Distinct from Ralph (single-session autonomous loop) — orchestrator coordinates multi-phase, multi-agent execution. See `docs/orchestrator-loop-architecture.md`.
|
||||
|
||||
**Hook events**: Claude Code hooks trigger via `/api/hook-event`. Key events: `permission_prompt`, `elicitation_dialog`, `idle_prompt`, `stop`, `teammate_idle`, `task_completed`. See `src/hooks-config.ts`.
|
||||
|
||||
**Agent Teams**: `TeamWatcher` polls `~/.claude/teams/`, matches to sessions via `leadSessionId`. Teammates are in-process threads appearing as subagents. Enable: `CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1`. See `agent-teams/`.
|
||||
**Agent Teams**: `TeamWatcher` polls `~/.claude/teams/`, matches to sessions via `leadSessionId`. Teammates are in-process threads appearing as subagents. Enable: `CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1`. See `docs/agent-teams/`.
|
||||
|
||||
**Circuit breaker**: Prevents respawn thrashing. States: `CLOSED` → `HALF_OPEN` → `OPEN`. Reset: `/api/sessions/:id/ralph-circuit-breaker/reset`.
|
||||
|
||||
@@ -143,19 +161,26 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
|
||||
### Frontend
|
||||
|
||||
Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. Load order: `constants.js`(1) → `mobile-handlers.js`(2) → `voice-input.js`(3) → `notification-manager.js`(4) → `keyboard-accessory.js`(5) → `input-cjk.js`(5.5) → `app.js`(6) → `terminal-ui.js`(7) → `respawn-ui.js`(8) → `ralph-panel.js`(9) → `settings-ui.js`(10) → `panels-ui.js`(11) → `session-ui.js`(12) → `ralph-wizard.js`(13) → `api-client.js`(14) → `subagent-windows.js`(15). `input-cjk.js` handles CJK IME composition via an always-visible textarea below the terminal (`window.cjkActive` blocks xterm's onData).
|
||||
Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. Load order: `constants.js`(1) → `mobile-handlers.js`(2) → `voice-input.js`(3) → `notification-manager.js`(4) → `keyboard-accessory.js`(5) → `input-cjk.js`(5.5) → `app.js`(6) → `terminal-ui.js`(7) → `respawn-ui.js`(8) → `ralph-panel.js`(9) → `orchestrator-panel.js`(9.5) → `settings-ui.js`(10) → `panels-ui.js`(11) → `session-ui.js`(12) → `ralph-wizard.js`(13) → `api-client.js`(14) → `subagent-windows.js`(15). `input-cjk.js` handles CJK IME composition via an always-visible textarea below the terminal (`window.cjkActive` blocks xterm's onData).
|
||||
|
||||
**Z-index layers**: subagent windows (1000), plan agents (1100), log viewers (2000), image popups (3000), local echo overlay (7).
|
||||
|
||||
**Multi-monitor button** (header, top-right; the notification bell it sits beside stays hidden — notifications live in Settings → Notifications). `app.launchMultiMonitor()` (in `panels-ui.js`) POSTs `/api/system/span-displays`, which spawns `scripts/span-codeman.sh` — a fresh, maximized browser `--app` window sized to the union of all displays (macOS; needs "Displays have separate Spaces" OFF). Supports the gesture layer's in-page floating session panels dragging across the physical monitor seam. **Opt-in:** hidden by default; enable under App Settings → Display → **Header Displays** ("Multi-monitor Button", `showMultiMonitorButton`). The button carries a `btn-multimonitor--hidden` class in the template; `renderIndexHtml` strips that class at render when the setting is on (a unique class token, not a brittle match on the aria-label/style copy), and `applyHeaderVisibilitySettings()` toggles the same class live on save. Solo (detached) windows hide it via `body.solo-mode`.
|
||||
|
||||
**Gesture control** (the camera hand-tracking overlay) is **opt-in, default OFF**, under App Settings → Display → **Input** (`gestureControlEnabled`). `CODEMAN_GESTURE=1` makes the feature *available* on the instance (CSP widening + `/gesture/` assets) and sets `window.__codemanGestureAvailable` (the Input section only shows when set); the overlay bundle is injected by `renderIndexHtml` **only when the setting is enabled**, so that method is `async` and reads `settings.json` via `readSettings(true)` — the `true` forces a **fresh** read (bypassing the 2s `_settingsCache`), because a post-save reload happens within that TTL and the cached value would otherwise render the pre-toggle state. Toggling the setting reloads the page (the bundle is render-injected).
|
||||
|
||||
**Respawn presets**: `solo-work` (3s/60min), `subagent-workflow` (45s/240min), `team-lead` (90s/480min), `ralph-todo` (8s/480min), `overnight-autonomous` (10s/480min).
|
||||
|
||||
**Keyboard shortcuts**: Escape (close), Ctrl+? (help), Ctrl+Enter (quick start), Ctrl+W (kill), Ctrl+Tab (next), Ctrl+K (kill all), Ctrl+L (clear), Ctrl+Shift+R (restore size), Ctrl/Cmd +/- (font).
|
||||
**Keyboard shortcuts**: Escape (close), Ctrl+? (help), Ctrl+W (kill), Ctrl+Tab (next), Alt+1-9 (switch tab), Ctrl+Shift+{/} (move tab left/right), Shift+Enter (newline), Ctrl+L (clear), Ctrl+Shift+R (restore size), Ctrl+Shift+V (voice input), Ctrl/Cmd +/- (font).
|
||||
|
||||
### Security
|
||||
|
||||
**Full model: [`docs/security-architecture.md`](docs/security-architecture.md)** — network binding, auth pipeline, the tunnel caveat, file-serving hardening, supply-chain, instance isolation, and recommended secure setups.
|
||||
|
||||
| Layer | Details |
|
||||
|-------|---------|
|
||||
| **Auth** | Optional HTTP Basic via `CODEMAN_USERNAME`/`CODEMAN_PASSWORD` env vars |
|
||||
| **Auth** | Optional HTTP Basic via `CODEMAN_USERNAME` (defaults to `admin`) / `CODEMAN_PASSWORD` env vars. Active only when `CODEMAN_PASSWORD` is set (`middleware/auth.ts`) |
|
||||
| **Network bind** | Defaults to `127.0.0.1` (loopback). A non-loopback bind (`--host`/`CODEMAN_HOST`) without `CODEMAN_PASSWORD` **starts but warns loudly** (0.9.0; was fail-closed in COD-29/#107). `--allow-unauthenticated-network` / `CODEMAN_ALLOW_UNAUTHENTICATED_NETWORK=1` acknowledges the warning. Classifier: `network-auth-policy.ts` |
|
||||
| **QR Auth** | Single-use 6-char tokens (60s TTL) for tunnel login. See `docs/qr-auth-plan.md` |
|
||||
| **Sessions** | 24h cookie (`codeman_session`), auto-extend, device context audit |
|
||||
| **Rate limit** | 10 failed auth/IP → 429 (15min decay). QR has separate limiter |
|
||||
@@ -166,11 +191,11 @@ Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. L
|
||||
|
||||
### SSE Event Registry
|
||||
|
||||
~106 event types in `src/web/sse-events.ts` (backend) and `SSE_EVENTS` in `constants.js` (frontend). Both must be kept in sync.
|
||||
~120 event types in `src/web/sse-events.ts` (backend) and `SSE_EVENTS` in `constants.js` (frontend). Both must be kept in sync.
|
||||
|
||||
### API Routes
|
||||
|
||||
~114 handlers across 13 route files in `src/web/routes/`: system (36), sessions (25), ralph (9), plan (8), respawn (7), cases (7), files (5), mux (5), scheduled (4), push (4), teams (2), hooks (1), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details.
|
||||
~130 handlers across 15 route files in `src/web/routes/`: system (37, incl. `POST /api/system/span-displays` → spawns `scripts/span-codeman.sh`), sessions (28), orchestrator (10), cases (9), ralph (9), plan (8), respawn (7), files (5), mux (5), push (4), scheduled (4), teams (2), hooks (1), clipboard (1), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details.
|
||||
|
||||
## Adding Features
|
||||
|
||||
@@ -192,22 +217,20 @@ All in `~/.codeman/`: `state.json` (sessions, settings, respawn), `mux-sessions.
|
||||
**CRITICAL: You are running inside a Codeman-managed tmux session.** Never run `npx vitest run` (full suite) — it spawns/kills tmux sessions and will crash your own session. Only run individual files:
|
||||
|
||||
```bash
|
||||
npx vitest run test/<specific-file>.test.ts # Single file (SAFE)
|
||||
npx vitest run -t "pattern" # By name (SAFE)
|
||||
# npx vitest run # DANGEROUS — DON'T DO THIS
|
||||
npm test -- test/<specific-file>.test.ts # Single file (SAFE, uses config/vitest.config.ts)
|
||||
npm test -- -t "pattern" # By name (SAFE)
|
||||
# npm test # DANGEROUS — runs full suite, DON'T DO THIS
|
||||
```
|
||||
|
||||
Raw `npx vitest` skips `config/vitest.config.ts`; always use `npm test --` or pass `--config config/vitest.config.ts`.
|
||||
|
||||
**Config**: Vitest with `globals: true`, `fileParallelism: false`. Timeout 30s, teardown 60s.
|
||||
|
||||
**Safety**: `test/setup.ts` snapshots pre-existing tmux sessions and never kills them. Only `registerTestTmuxSession()` sessions get cleaned up.
|
||||
|
||||
**Ports**: Pick unique ports manually. Search `const PORT =` before adding new tests.
|
||||
|
||||
**Respawn tests**: Use `MockSession` from `test/respawn-test-utils.ts`. **Route tests**: `app.inject()` in `test/routes/`. **Mobile tests**: Playwright suite in `mobile-test/` (135 device profiles).
|
||||
|
||||
## Screenshots
|
||||
|
||||
Mobile screenshots in `~/.codeman/screenshots/`. API: `GET /api/screenshots`, `POST /api/screenshots`.
|
||||
**Respawn tests**: Use `MockSession` from `test/respawn-test-utils.ts`. **Route tests**: `app.inject({ method, url, payload })` in `test/routes/` — no live port needed. **Mobile tests**: Playwright suite in `test/mobile/` (135 device profiles).
|
||||
|
||||
## Debugging
|
||||
|
||||
@@ -219,27 +242,14 @@ curl localhost:3000/api/subagents | jq # Background agents
|
||||
cat ~/.codeman/state.json | jq # Persisted state
|
||||
```
|
||||
|
||||
Mobile screenshots: `~/.codeman/screenshots/`, accessed via `GET/POST /api/screenshots`.
|
||||
|
||||
## Performance & Limits
|
||||
|
||||
Target: 20 sessions, 50 agent windows at 60fps. Limits in `src/config/`: terminal 2MB, text 1MB, messages 1000, max agents 500, max sessions 50, max SSE clients 100. Use `LRUMap` for bounded caches, `StaleExpirationMap` for TTL cleanup. Anti-flicker pipeline: `docs/terminal-anti-flicker.md`.
|
||||
|
||||
## References
|
||||
**Memory leaks (24+ hour sessions)**: use `CleanupManager`, clear Maps in `stop()`, guard async with `if (this.cleanup.isStopped) return`. Frontend: store handler refs, clean in `close*()`. Verify: `npm test -- test/memory-leak-prevention.test.ts`.
|
||||
|
||||
Deep-dive docs in `docs/`: `respawn-state-machine.md`, `ralph-wiggum-guide.md`, `claude-code-hooks-reference.md`, `terminal-anti-flicker.md`, `opencode-integration.md`, `qr-auth-plan.md`. Agent Teams: `agent-teams/README.md`. SSE events: `src/web/sse-events.ts` + `constants.js`.
|
||||
## Scripts & Tunnel
|
||||
|
||||
## Scripts
|
||||
|
||||
Key: `scripts/tmux-manager.sh` (safe tmux mgmt), `scripts/tunnel.sh` (tunnel start/stop/url). Production: `scripts/codeman-web.service`, `scripts/codeman-tunnel.service`.
|
||||
|
||||
## Memory Leak Prevention
|
||||
|
||||
24+ hour sessions: use `CleanupManager`, clear Maps in `stop()`, guard async with `if (this.cleanup.isStopped) return`. Frontend: store handler refs, clean in `close*()`. Verify: `npx vitest run test/memory-leak-prevention.test.ts`.
|
||||
|
||||
## Common Workflows
|
||||
|
||||
**Bug investigation**: Dev server → reproduce in browser → check terminal + `~/.codeman/state.json`.
|
||||
**Respawn changes**: Read `docs/respawn-state-machine.md` first. Use `MockSession` from `test/respawn-test-utils.ts`.
|
||||
|
||||
## Tunnel
|
||||
|
||||
`./scripts/tunnel.sh start|stop|url`. **Always set `CODEMAN_PASSWORD`** before exposing via tunnel.
|
||||
Key scripts: `scripts/tmux-manager.sh` (safe tmux mgmt), `scripts/tunnel.sh start|stop|url` (tunnel). Production services: `scripts/codeman-web.service`, `scripts/codeman-tunnel.service`. **Always set `CODEMAN_PASSWORD`** before exposing via tunnel.
|
||||
|
||||
@@ -30,24 +30,6 @@ curl -fsSL https://raw.githubusercontent.com/Ark0N/Codeman/master/install.sh | b
|
||||
|
||||
This installs Node.js and tmux if missing, clones Codeman to `~/.codeman/app`, and builds it.
|
||||
|
||||
**Install from a fork or specific branch:**
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/<user>/Codeman/<branch>/install.sh | \
|
||||
CODEMAN_REPO_URL=https://github.com/<user>/Codeman.git \
|
||||
CODEMAN_BRANCH=<branch> bash
|
||||
```
|
||||
|
||||
The installer supports these environment variables:
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `CODEMAN_REPO_URL` | upstream Codeman | Custom git repository URL |
|
||||
| `CODEMAN_BRANCH` | `master` | Git branch to install |
|
||||
| `CODEMAN_INSTALL_DIR` | `~/.codeman/app` | Custom install directory |
|
||||
| `CODEMAN_SKIP_SYSTEMD` | `0` | Skip systemd service setup prompt |
|
||||
| `CODEMAN_NODE_VERSION` | `22` | Node.js major version to install |
|
||||
| `CODEMAN_NONINTERACTIVE` | `0` | Skip all prompts (for CI/automation) |
|
||||
|
||||
You'll need at least one AI coding CLI installed — [Claude Code](https://docs.anthropic.com/en/docs/claude-code) or [OpenCode](https://opencode.ai) (or both). After install:
|
||||
|
||||
```bash
|
||||
@@ -60,12 +42,53 @@ codeman web
|
||||
|
||||
**Linux (systemd):**
|
||||
```bash
|
||||
mkdir -p ~/.config/systemd/user && printf '[Unit]\nDescription=Codeman Web Server\nAfter=network.target\n\n[Service]\nType=simple\nExecStart=%s %s/dist/index.js web\nRestart=always\nRestartSec=10\n\n[Install]\nWantedBy=default.target\n' "$(which node)" "$HOME/.codeman/app" > ~/.config/systemd/user/codeman-web.service && systemctl --user daemon-reload && systemctl --user enable --now codeman-web && loginctl enable-linger $USER
|
||||
mkdir -p ~/.config/systemd/user
|
||||
cat > ~/.config/systemd/user/codeman-web.service << EOF
|
||||
[Unit]
|
||||
Description=Codeman Web Server
|
||||
After=network.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
ExecStart=$(which node) $HOME/.codeman/app/dist/index.js web
|
||||
Restart=always
|
||||
RestartSec=10
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
EOF
|
||||
systemctl --user daemon-reload
|
||||
systemctl --user enable --now codeman-web
|
||||
loginctl enable-linger $USER
|
||||
```
|
||||
|
||||
**macOS (launchd):**
|
||||
```bash
|
||||
mkdir -p ~/Library/LaunchAgents && printf '<?xml version="1.0" encoding="UTF-8"?>\n<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">\n<plist version="1.0"><dict><key>Label</key><string>com.codeman.web</string><key>ProgramArguments</key><array><string>%s</string><string>%s/dist/index.js</string><string>web</string></array><key>RunAtLoad</key><true/><key>KeepAlive</key><true/><key>StandardOutPath</key><string>/tmp/codeman.log</string><key>StandardErrorPath</key><string>/tmp/codeman.log</string></dict></plist>\n' "$(which node)" "$HOME/.codeman/app" > ~/Library/LaunchAgents/com.codeman.web.plist && launchctl load ~/Library/LaunchAgents/com.codeman.web.plist
|
||||
mkdir -p ~/Library/LaunchAgents
|
||||
cat > ~/Library/LaunchAgents/com.codeman.web.plist << EOF
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN"
|
||||
"http://www.apple.com/DTDs/PropertyList-1.0.dtd">
|
||||
<plist version="1.0">
|
||||
<dict>
|
||||
<key>Label</key>
|
||||
<string>com.codeman.web</string>
|
||||
<key>ProgramArguments</key>
|
||||
<array>
|
||||
<string>$(which node)</string>
|
||||
<string>$HOME/.codeman/app/dist/index.js</string>
|
||||
<string>web</string>
|
||||
</array>
|
||||
<key>RunAtLoad</key><true/>
|
||||
<key>KeepAlive</key><true/>
|
||||
<key>StandardOutPath</key>
|
||||
<string>/tmp/codeman.log</string>
|
||||
<key>StandardErrorPath</key>
|
||||
<string>/tmp/codeman.log</string>
|
||||
</dict>
|
||||
</plist>
|
||||
EOF
|
||||
launchctl bootstrap gui/$(id -u) ~/Library/LaunchAgents/com.codeman.web.plist
|
||||
```
|
||||
</details>
|
||||
|
||||
@@ -203,6 +226,17 @@ Run **20 parallel sessions** with full visibility — real-time xterm.js termina
|
||||
|
||||
Every session runs inside **tmux** — sessions survive server restarts, network drops, and machine sleep. Auto-recovery on startup with dual redundancy. Ghost session discovery finds orphaned tmux sessions. Managed sessions are environment-tagged so the agent won't kill its own session.
|
||||
|
||||
### Hostname-Aware Window Title
|
||||
|
||||
Running Codeman on multiple hosts (laptop, dev box, NAS)? The browser tab title is `codeman:<hostname>` so you can tell which backend each tab points at without clicking in:
|
||||
|
||||
```bash
|
||||
codeman web # codeman:<os.hostname()>
|
||||
codeman web --title-hostname dev-box # codeman:dev-box (manual override for noisy hostnames)
|
||||
```
|
||||
|
||||
The title is templated into the served HTML on first byte, so it's correct from the very first paint and works without JavaScript. The same hostname prefix is applied to the tab-flash format (`⚠️ (N) codeman:<host>`) and to OS-level desktop notifications (`codeman:<host>: <event>`), so cross-host alerts in the system notification center are also unambiguous.
|
||||
|
||||
### Smart Token Management
|
||||
|
||||
| Threshold | Action | Result |
|
||||
@@ -378,9 +412,12 @@ Single-digit selection (1-9), color-coded status, token counts, auto-refresh. De
|
||||
| `Ctrl+Enter` | Quick-start session |
|
||||
| `Ctrl+W` | Close session |
|
||||
| `Ctrl+Tab` | Next session |
|
||||
| `Alt+1`–`Alt+9` | Switch to tab N |
|
||||
| `Ctrl+Shift+{` / `Ctrl+Shift+}` | Move active tab left / right |
|
||||
| `Ctrl+K` | Kill all sessions |
|
||||
| `Ctrl+L` | Clear terminal |
|
||||
| `Ctrl+Shift+R` | Restore terminal size |
|
||||
| `Ctrl+Shift+V` | Toggle voice input |
|
||||
| `Ctrl/Cmd +/-` | Font size |
|
||||
| `Escape` | Close panels |
|
||||
|
||||
@@ -423,6 +460,7 @@ Single-digit selection (1-9), color-coded status, token counts, auto-refresh. De
|
||||
| `GET` | `/api/events` | SSE stream |
|
||||
| `GET` | `/api/status` | Full app state |
|
||||
| `POST` | `/api/hook-event` | Hook callbacks |
|
||||
| `POST` | `/api/clipboard` | Push text to all connected browsers (`{text}`) |
|
||||
| `GET` | `/api/sessions/:id/run-summary` | Timeline + stats |
|
||||
|
||||
---
|
||||
|
||||
@@ -22,8 +22,7 @@ export default tseslint.config(
|
||||
'src/web/public/vendor/**',
|
||||
'src/web/public/app.js',
|
||||
'scripts/**/*.mjs',
|
||||
'tools/**',
|
||||
'remotion/**',
|
||||
'scripts/remotion/**',
|
||||
],
|
||||
}
|
||||
);
|
||||
@@ -1,7 +1,11 @@
|
||||
import { resolve } from 'node:path';
|
||||
import { defineConfig } from 'vitest/config';
|
||||
|
||||
const root = resolve(import.meta.dirname, '..');
|
||||
|
||||
export default defineConfig({
|
||||
test: {
|
||||
root,
|
||||
globals: true,
|
||||
environment: 'node',
|
||||
include: ['test/**/*.test.ts'],
|
||||
@@ -1,3 +1,9 @@
|
||||
> **⚠️ ARCHIVED 2026-05-21 — superseded, kept for history.**
|
||||
> The headline items here were verified resolved: the P0 `{WORKING_DIR}` placeholder
|
||||
> is now replaced (`plan-orchestrator.ts:431`), and the "~66 dead functions in app.js"
|
||||
> are gone (app.js was modularized 15K→3K LOC). A fresh `npm run knip` sweep on
|
||||
> 2026-05-21 found only a handful of unused test helpers. Do not treat this as a live TODO.
|
||||
|
||||
# Codebase Cleanup Findings
|
||||
|
||||
Compiled from parallel analysis of the entire Codeman codebase by 3 research agents (2026-02-19).
|
||||
@@ -1,3 +1,9 @@
|
||||
> **⚠️ ARCHIVED 2026-05-21 — superseded, kept for history.**
|
||||
> The "Critical" structural items here are done: `server.ts` 6,736→2,065 LOC,
|
||||
> `app.js` 15,196→3,083 LOC, `types.ts` 1,443→12 LOC (now a barrel → `src/types/`).
|
||||
> The phase plans that executed this work are in `docs/archive/phase*-plan.md`.
|
||||
> Do not treat this as a live TODO; see CLAUDE.md for current architecture.
|
||||
|
||||
# Code Structure & Quality Findings
|
||||
|
||||
**Date**: 2026-02-28
|
||||
@@ -1,5 +1,7 @@
|
||||
# Local Echo Overlay — Implementation Plan
|
||||
|
||||
> **Status: SHIPPED.** Implementation lives in `packages/xterm-zerolag-input/src/` (overlay-renderer.ts, prompt-finder.ts, cell-dimensions.ts, zerolag-input-addon.ts) with the embedded copy in `src/web/public/app.js`. This document is retained as historical design context.
|
||||
|
||||
## Context
|
||||
|
||||
User accesses Codeman remotely from Thailand to Switzerland over Tailscale (~200-300ms RTT).
|
||||
@@ -18,9 +20,9 @@ redraws. A DOM overlay sits in a separate rendering layer (z-index 7) and doesn'
|
||||
with Ink's cursor management or screen redraws at all. When Ink redraws (server output arrives),
|
||||
we simply hide the overlay.
|
||||
|
||||
**Why it will look indistinguishable:** We use the DOM renderer (not canvas/WebGL) in our
|
||||
xterm.js v5.3.0, so both terminal text and overlay text are rendered by the same browser
|
||||
font engine with identical sub-pixel rendering.
|
||||
**Why it will look indistinguishable:** We use the DOM renderer (not canvas/WebGL), so both
|
||||
terminal text and overlay text are rendered by the same browser font engine with identical
|
||||
sub-pixel rendering. (Originally designed against xterm.js v5.3.0; project now on `@xterm/xterm` ^6.0.0 — the internal `_core._renderService.dimensions` access path still works in v6.)
|
||||
|
||||
## Key Technical Details (from research)
|
||||
|
||||
@@ -36,7 +38,7 @@ const top = cursorY * dims.css.cell.height; // CSS pixels, relative to .xterm-
|
||||
- `cursorY` = `terminal.buffer.active.cursorY` (0 to terminal.rows-1, ALREADY viewport-relative)
|
||||
- No scroll offset math needed
|
||||
|
||||
### Cell Dimensions (v5.3.0 — no public API, use internal)
|
||||
### Cell Dimensions (no public API in v5/v6 — use internal; public in v7+)
|
||||
```js
|
||||
const dims = terminal._core._renderService.dimensions;
|
||||
dims.css.cell.width // e.g., 8.4px
|
||||
|
||||
@@ -0,0 +1,367 @@
|
||||
# Orchestrator Loop — Architecture & Data Flow
|
||||
|
||||
> Technical architecture document. Not for GitHub.
|
||||
|
||||
## System Overview
|
||||
|
||||
```
|
||||
┌─────────────────────────────────────────────────────────────────────┐
|
||||
│ CODEMAN WEB UI │
|
||||
│ ┌──────────────────────────────────────────────────────────────┐ │
|
||||
│ │ Orchestrator Dashboard │ │
|
||||
│ │ [Goal Input] [Plan View] [Phase Progress] [Agent Activity] │ │
|
||||
│ └───────────────────────────┬──────────────────────────────────┘ │
|
||||
│ │ SSE Events │
|
||||
│ ▼ │
|
||||
│ ┌──────────────────────────────────────────────────────────────┐ │
|
||||
│ │ Orchestrator API Routes (/api/orchestrator/*) │ │
|
||||
│ └───────────────────────────┬──────────────────────────────────┘ │
|
||||
└───────────────────────────────┼─────────────────────────────────────┘
|
||||
▼
|
||||
┌─────────────────────────────────────────────────────────────────────┐
|
||||
│ ORCHESTRATOR LOOP │
|
||||
│ │
|
||||
│ ┌──────────────┐ ┌──────────────┐ ┌──────────────────────┐ │
|
||||
│ │ Orchestrator │ │ Orchestrator │ │ Orchestrator │ │
|
||||
│ │ Planner │ │ Loop (state │ │ Verifier │ │
|
||||
│ │ │ │ machine) │ │ │ │
|
||||
│ │ • Research │◄──►│ • Phase mgmt │◄──►│ • Test runner │ │
|
||||
│ │ • Plan gen │ │ • Task queue │ │ • AI review │ │
|
||||
│ │ • Phasing │ │ • Event loop │ │ • Output checks │ │
|
||||
│ └──────┬───────┘ └──────┬───────┘ └──────────┬───────────┘ │
|
||||
│ │ │ │ │
|
||||
│ ▼ ▼ ▼ │
|
||||
│ ┌──────────────────────────────────────────────────────────────┐ │
|
||||
│ │ EXISTING CODEMAN INFRASTRUCTURE │ │
|
||||
│ │ │ │
|
||||
│ │ SessionManager ←→ Sessions ←→ PTY (Claude CLI) │ │
|
||||
│ │ ↑ ↑ ↑ │ │
|
||||
│ │ │ │ │ │ │
|
||||
│ │ TaskQueue RalphTracker RespawnController │ │
|
||||
│ │ StateStore HooksConfig TeamWatcher │ │
|
||||
│ │ Auto-Ops SubagentWatcher SSE Broadcast │ │
|
||||
│ └──────────────────────────────────────────────────────────────┘ │
|
||||
└─────────────────────────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
## Data Flow: Complete Lifecycle
|
||||
|
||||
### 1. User Submits Goal
|
||||
|
||||
```
|
||||
User → POST /api/orchestrator/start { goal: "Build a REST API...", config: {...} }
|
||||
→ OrchestratorLoop.start(goal)
|
||||
→ state = PLANNING
|
||||
→ emit('stateChanged', 'planning')
|
||||
→ SSE: orchestrator:stateChanged
|
||||
```
|
||||
|
||||
### 2. Planning Phase
|
||||
|
||||
```
|
||||
OrchestratorPlanner.generatePlan(goal)
|
||||
→ PlanOrchestrator.generateDetailedPlan(goal)
|
||||
→ [Research Agent] → enriched task description
|
||||
→ [Planner Agent] → PlanItem[]
|
||||
→ groupIntoPhases(planItems)
|
||||
→ topological sort by dependencies
|
||||
→ group into layers
|
||||
→ assign team strategies
|
||||
→ OrchestratorPlan { phases: [...] }
|
||||
→ state = APPROVAL
|
||||
→ emit('planReady', plan)
|
||||
→ SSE: orchestrator:planReady
|
||||
```
|
||||
|
||||
### 3. User Approves Plan
|
||||
|
||||
```
|
||||
User → POST /api/orchestrator/approve
|
||||
→ OrchestratorLoop.approvePlan()
|
||||
→ state = EXECUTING
|
||||
→ executePhase(phases[0])
|
||||
```
|
||||
|
||||
### 4. Phase Execution
|
||||
|
||||
```
|
||||
executePhase(phase)
|
||||
→ For each task in phase:
|
||||
→ Convert to CreateTaskOptions
|
||||
→ Add to TaskQueue with completion phrase "PHASE_{N}_TASK_{M}_DONE"
|
||||
→ If phase.teamStrategy.type === 'team':
|
||||
→ Start session with AGENT_TEAMS enabled
|
||||
→ Send team orchestration prompt to lead
|
||||
→ Else:
|
||||
→ Assign tasks to available sessions (same as RalphLoop)
|
||||
|
||||
→ Listen for task completion events:
|
||||
→ TaskQueue emits taskCompleted
|
||||
→ Check: all phase tasks done?
|
||||
→ Yes → state = VERIFYING → verifyPhase(phase)
|
||||
→ No → wait for more completions
|
||||
```
|
||||
|
||||
### 5. Verification
|
||||
|
||||
```
|
||||
verifyPhase(phase)
|
||||
→ OrchestratorVerifier.verify(phase, session)
|
||||
→ Run test commands via session
|
||||
→ Check file existence
|
||||
→ AI review (optional)
|
||||
→ If passed:
|
||||
→ phase.status = 'passed'
|
||||
→ emit('phaseCompleted', phase)
|
||||
→ If more phases: executePhase(nextPhase)
|
||||
→ If last phase: state = COMPLETED
|
||||
→ If failed:
|
||||
→ phase.attempts++
|
||||
→ If attempts < maxAttempts:
|
||||
→ state = REPLANNING
|
||||
→ Generate recovery tasks
|
||||
→ state = EXECUTING (retry)
|
||||
→ Else:
|
||||
→ state = FAILED
|
||||
→ emit('phaseFailed', phase, reason)
|
||||
```
|
||||
|
||||
### 6. Context Management Between Phases
|
||||
|
||||
```
|
||||
After phase completion:
|
||||
→ If config.compactBetweenPhases:
|
||||
→ session.sendInput('/compact')
|
||||
→ Wait for compact to complete
|
||||
→ If config.respawnBetweenMilestones && phase is a milestone:
|
||||
→ Save orchestrator state to StateStore
|
||||
→ Respawn session (kill + recreate)
|
||||
→ Send resume prompt with phase context
|
||||
```
|
||||
|
||||
## File Layout
|
||||
|
||||
```
|
||||
src/
|
||||
├── orchestrator-loop.ts # Main state machine (~400 lines)
|
||||
├── orchestrator-planner.ts # Plan generation + phase grouping (~300 lines)
|
||||
├── orchestrator-verifier.ts # Phase verification (~200 lines)
|
||||
├── types/
|
||||
│ └── orchestrator.ts # All orchestrator types (~150 lines)
|
||||
├── prompts/
|
||||
│ └── orchestrator.ts # Prompt templates (~200 lines)
|
||||
├── web/
|
||||
│ ├── routes/
|
||||
│ │ └── orchestrator-routes.ts # API endpoints (~250 lines)
|
||||
│ └── public/
|
||||
│ └── orchestrator-ui.js # Frontend panel (~500 lines)
|
||||
```
|
||||
|
||||
## Integration Points with Existing Code
|
||||
|
||||
### StateStore (`src/state-store.ts`)
|
||||
```typescript
|
||||
// Add to AppState interface
|
||||
orchestrator?: OrchestratorPersistState;
|
||||
|
||||
// Add methods
|
||||
getOrchestratorState(): OrchestratorPersistState;
|
||||
setOrchestratorState(state: Partial<OrchestratorPersistState>): void;
|
||||
```
|
||||
|
||||
### SSE Events (`src/web/sse-events.ts`)
|
||||
```typescript
|
||||
// Add ~8 new events
|
||||
export const SseEvent = {
|
||||
// ... existing
|
||||
ORCHESTRATOR_STATE_CHANGED: 'orchestrator:stateChanged',
|
||||
ORCHESTRATOR_PLAN_READY: 'orchestrator:planReady',
|
||||
ORCHESTRATOR_PHASE_STARTED: 'orchestrator:phaseStarted',
|
||||
ORCHESTRATOR_PHASE_COMPLETED: 'orchestrator:phaseCompleted',
|
||||
ORCHESTRATOR_PHASE_FAILED: 'orchestrator:phaseFailed',
|
||||
ORCHESTRATOR_VERIFICATION: 'orchestrator:verificationResult',
|
||||
ORCHESTRATOR_COMPLETED: 'orchestrator:completed',
|
||||
ORCHESTRATOR_ERROR: 'orchestrator:error',
|
||||
} as const;
|
||||
```
|
||||
|
||||
### Frontend Constants (`src/web/public/constants.js`)
|
||||
```javascript
|
||||
// Mirror SSE events
|
||||
SSE_EVENTS.ORCHESTRATOR_STATE_CHANGED = 'orchestrator:stateChanged';
|
||||
// ... etc
|
||||
```
|
||||
|
||||
### Route Registration (`src/web/routes/index.ts`)
|
||||
```typescript
|
||||
import { registerOrchestratorRoutes } from './orchestrator-routes.js';
|
||||
// Add to barrel export
|
||||
```
|
||||
|
||||
### Server (`src/web/server.ts`)
|
||||
```typescript
|
||||
// Initialize OrchestratorLoop alongside RalphLoop
|
||||
const orchestratorLoop = new OrchestratorLoop(config);
|
||||
|
||||
// Register routes
|
||||
registerOrchestratorRoutes(app, { ...ctx, orchestrator: orchestratorLoop });
|
||||
```
|
||||
|
||||
### Port Interface (`src/web/ports/`)
|
||||
```typescript
|
||||
// New port
|
||||
export interface OrchestratorPort {
|
||||
orchestrator: OrchestratorLoop;
|
||||
}
|
||||
```
|
||||
|
||||
## Prompt Flow Through System
|
||||
|
||||
The key insight is how prompts flow from Orchestrator → Session → Claude:
|
||||
|
||||
```
|
||||
OrchestratorLoop decides to execute Phase 3, Task 2
|
||||
│
|
||||
▼
|
||||
Converts OrchestratorTask to CreateTaskOptions:
|
||||
{
|
||||
prompt: "Implement the rate limiter middleware. Read src/middleware/auth.ts
|
||||
for the pattern. Add to src/middleware/rate-limiter.ts. Must export
|
||||
a Fastify plugin. When done: <promise>PHASE_3_TASK_2_DONE</promise>",
|
||||
priority: 100,
|
||||
dependencies: ["phase-3-task-1"], // Must finish auth middleware first
|
||||
completionPhrase: "PHASE_3_TASK_2_DONE",
|
||||
timeoutMs: 600000 // 10 minutes
|
||||
}
|
||||
│
|
||||
▼
|
||||
TaskQueue.addTask(options)
|
||||
│
|
||||
▼
|
||||
RalphLoop.tick() → assignTasks() // OR OrchestratorLoop does its own assignment
|
||||
│
|
||||
▼
|
||||
session.sendInput(task.prompt)
|
||||
│
|
||||
▼
|
||||
writeViaMux() → tmux send-keys -l "prompt..." + Enter
|
||||
│
|
||||
▼
|
||||
Claude CLI receives prompt, executes, outputs results
|
||||
│
|
||||
▼
|
||||
RalphTracker.processData() → detects "PHASE_3_TASK_2_DONE"
|
||||
│
|
||||
▼
|
||||
emit('completionDetected') → OrchestratorLoop.handleTaskCompleted()
|
||||
│
|
||||
▼
|
||||
Check: all tasks in Phase 3 done? → If yes → verifyPhase(phase3)
|
||||
```
|
||||
|
||||
## Team Agent Flow (When Enabled)
|
||||
|
||||
```
|
||||
Phase has teamStrategy.type === 'team'
|
||||
│
|
||||
▼
|
||||
OrchestratorLoop creates/reuses a session with:
|
||||
env: { CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS: '1' }
|
||||
│
|
||||
▼
|
||||
Sends team orchestration prompt:
|
||||
"You're the team lead for Phase 3: Core Implementation.
|
||||
|
||||
Your team should work on these tasks in parallel:
|
||||
1. Rate limiter middleware (teammate 1)
|
||||
2. Error handling middleware (teammate 2)
|
||||
3. Validation layer (teammate 3)
|
||||
|
||||
Context files to read first: [...]
|
||||
Each teammate should output their task's completion phrase when done.
|
||||
When ALL tasks are complete, output: <promise>PHASE_3_COMPLETE</promise>"
|
||||
│
|
||||
▼
|
||||
Claude Code team-lead spawns teammates
|
||||
│
|
||||
▼
|
||||
TeamWatcher detects new team in ~/.claude/teams/
|
||||
→ Matches to session via leadSessionId
|
||||
→ Tracks teammate activity
|
||||
│
|
||||
▼
|
||||
Teammates work in parallel (in-process threads)
|
||||
│
|
||||
▼
|
||||
hook: teammate_idle → POST /api/hook-event
|
||||
→ OrchestratorLoop notes teammate finished
|
||||
│
|
||||
▼
|
||||
hook: task_completed → POST /api/hook-event
|
||||
→ Or: RalphTracker detects PHASE_3_COMPLETE
|
||||
→ OrchestratorLoop → phase complete → verify
|
||||
```
|
||||
|
||||
## Error Recovery Strategy
|
||||
|
||||
```
|
||||
Task fails (timeout, error, session crash)
|
||||
│
|
||||
├─ Task-level retry (up to 2 retries per task)
|
||||
│ → Reset task to pending
|
||||
│ → Re-queue with modified prompt: "Previous attempt failed: {error}. Try again..."
|
||||
│
|
||||
├─ Phase-level retry (up to 3 retries per phase)
|
||||
│ → Respawn session (fresh context)
|
||||
│ → Re-execute entire phase with learnings from failure
|
||||
│ → Modified prompt includes what went wrong
|
||||
│
|
||||
└─ Orchestration-level failure
|
||||
→ All retries exhausted
|
||||
→ state = FAILED
|
||||
→ Notify user with detailed failure report
|
||||
→ User can: modify plan → retry, skip phase → continue, or stop
|
||||
```
|
||||
|
||||
## Interaction with Ralph Loop
|
||||
|
||||
Ralph Loop and Orchestrator Loop are **mutually exclusive** on the same sessions:
|
||||
|
||||
```
|
||||
if (orchestratorLoop.isRunning()) {
|
||||
// Orchestrator controls task assignment
|
||||
// Ralph Loop should not interfere
|
||||
// Respawn Controller uses 'orchestrator' preset
|
||||
}
|
||||
|
||||
if (ralphLoop.isRunning()) {
|
||||
// Ralph controls task assignment
|
||||
// Orchestrator should not start
|
||||
}
|
||||
```
|
||||
|
||||
The Orchestrator can optionally USE the Ralph Loop internally for phase execution (delegate phase tasks to Ralph's queue), or manage task assignment directly. Decision: **manage directly** — gives more control over phase boundaries and verification timing.
|
||||
|
||||
## Summary of What Touches What
|
||||
|
||||
| Existing File | Change |
|
||||
|---|---|
|
||||
| `src/types/index.ts` | Export orchestrator types |
|
||||
| `src/state-store.ts` | Add orchestrator state persistence |
|
||||
| `src/web/sse-events.ts` | Add ~8 orchestrator events |
|
||||
| `src/web/routes/index.ts` | Register orchestrator routes |
|
||||
| `src/web/server.ts` | Initialize OrchestratorLoop |
|
||||
| `src/web/public/constants.js` | Mirror SSE events |
|
||||
| `src/web/public/app.js` | Add orchestrator event listeners, panel toggle |
|
||||
| `src/web/route-helpers.ts` | Add 'orchestrator' respawn preset |
|
||||
|
||||
| New File | Purpose |
|
||||
|---|---|
|
||||
| `src/orchestrator-loop.ts` | Core state machine |
|
||||
| `src/orchestrator-planner.ts` | Plan generation + phasing |
|
||||
| `src/orchestrator-verifier.ts` | Phase verification |
|
||||
| `src/types/orchestrator.ts` | Type definitions |
|
||||
| `src/prompts/orchestrator.ts` | Prompt templates |
|
||||
| `src/web/routes/orchestrator-routes.ts` | API endpoints |
|
||||
| `src/web/public/orchestrator-ui.js` | Frontend panel |
|
||||
| `src/web/ports/orchestrator-port.ts` | Port interface |
|
||||
@@ -0,0 +1,633 @@
|
||||
# Orchestrator Loop — Detailed Implementation Plan (v2)
|
||||
|
||||
> Internal research/planning document. Not for GitHub.
|
||||
|
||||
## Vision
|
||||
|
||||
The **Orchestrator Loop** is a new autonomous execution mode that transforms high-level user goals into phased, verified, team-coordinated implementations. Unlike Ralph Loop (flat task queue → idle sessions), the Orchestrator manages the full lifecycle: **plan → approve → execute → verify → adapt → complete**.
|
||||
|
||||
```
|
||||
USER: "Add OAuth2 login with Google/GitHub, role-based access control, and API key management"
|
||||
|
||||
ORCHESTRATOR:
|
||||
Phase 1: Research & Setup ✅ (3m) — scaffold, deps, config
|
||||
Phase 2: Auth Core ✅ (8m) — OAuth2 flow, session mgmt
|
||||
Phase 3: Provider Integration 🔄 (12m) — Google + GitHub (parallel via team agents)
|
||||
Phase 4: RBAC ⏳ — roles, permissions, middleware
|
||||
Phase 5: API Keys ⏳ — generation, validation, rate limits
|
||||
Phase 6: Testing & Review ⏳ — integration tests, security review
|
||||
|
||||
Progress: ━━━━━━━━━━━━━━━━━━━━ 40% | Agents: 3 active | Time: 23m
|
||||
```
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
┌─────────────────────────────────────────────────────────────────┐
|
||||
│ OrchestratorLoop │
|
||||
│ │
|
||||
│ ┌────────────────┐ ┌────────────────┐ ┌──────────────────┐ │
|
||||
│ │ Orchestrator │ │ Orchestrator │ │ Orchestrator │ │
|
||||
│ │ Planner │ │ Executor │ │ Verifier │ │
|
||||
│ │ │ │ │ │ │ │
|
||||
│ │ PlanOrchestrator│ │ TaskQueue │ │ AI review │ │
|
||||
│ │ + phase grouper│ │ SessionManager │ │ Test commands │ │
|
||||
│ │ + team strategy│ │ Team prompts │ │ File checks │ │
|
||||
│ └───────┬────────┘ └───────┬────────┘ └─────────┬────────┘ │
|
||||
│ │ │ │ │
|
||||
│ └───────────────────┼──────────────────────┘ │
|
||||
│ │ │
|
||||
│ ┌─────────▼─────────┐ │
|
||||
│ │ Existing Codeman │ │
|
||||
│ │ Infrastructure │ │
|
||||
│ │ │ │
|
||||
│ │ SessionManager │ │
|
||||
│ │ TaskQueue │ │
|
||||
│ │ RespawnController │ │
|
||||
│ │ TeamWatcher │ │
|
||||
│ │ PlanOrchestrator │ │
|
||||
│ │ StateStore │ │
|
||||
│ │ Hooks + SSE │ │
|
||||
│ └────────────────────┘ │
|
||||
└─────────────────────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
## State Machine
|
||||
|
||||
```
|
||||
┌─────────┐
|
||||
│ IDLE │
|
||||
└────┬────┘
|
||||
│ start(goal)
|
||||
▼
|
||||
┌─────────┐
|
||||
┌────────│PLANNING │────────┐
|
||||
│ fail └────┬────┘ │
|
||||
▼ │ plan ready │ user cancels
|
||||
┌────────┐ ▼ ▼
|
||||
│ FAILED │ ┌─────────┐ ┌────────┐
|
||||
└────────┘ │APPROVAL │ │ IDLE │
|
||||
▲ └────┬────┘ └────────┘
|
||||
│ │ approve
|
||||
│ ▼
|
||||
│ ┌──────────┐
|
||||
│ ┌───►│EXECUTING │◄────────────────────┐
|
||||
│ │ └────┬─────┘ │
|
||||
│ │ │ all tasks in phase done │
|
||||
│ │ ▼ │
|
||||
│ │ ┌──────────┐ │
|
||||
│ │ │VERIFYING │ │
|
||||
│ │ └────┬─────┘ │
|
||||
│ │ pass │ │ fail │
|
||||
│ │ ▼ ▼ │
|
||||
│ │ more ┌──────────┐ │
|
||||
│ │ phases?│REPLANNING│── retry ────────┘
|
||||
│ │ │ └────┬─────┘
|
||||
│ │ │ │ max retries
|
||||
│ │ │ ▼
|
||||
│ │ │ ┌────────┐
|
||||
│ └────┘ │ FAILED │
|
||||
│ next └────────┘
|
||||
│ phase
|
||||
│ │
|
||||
│ ▼
|
||||
│ ┌───────────┐
|
||||
└─│ COMPLETED │
|
||||
└───────────┘
|
||||
```
|
||||
|
||||
**States:** `idle` | `planning` | `approval` | `executing` | `verifying` | `replanning` | `completed` | `failed` | `paused`
|
||||
|
||||
Transitions are event-driven. The state machine is the single source of truth — all methods check `this.state` before acting.
|
||||
|
||||
## Type Definitions
|
||||
|
||||
### `src/types/orchestrator.ts`
|
||||
|
||||
```typescript
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// State Machine
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
export type OrchestratorState =
|
||||
| 'idle'
|
||||
| 'planning'
|
||||
| 'approval'
|
||||
| 'executing'
|
||||
| 'verifying'
|
||||
| 'replanning'
|
||||
| 'completed'
|
||||
| 'failed'
|
||||
| 'paused';
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Plan Structure
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
export interface OrchestratorPlan {
|
||||
id: string;
|
||||
goal: string;
|
||||
createdAt: number;
|
||||
phases: OrchestratorPhase[];
|
||||
metadata: {
|
||||
totalTasks: number;
|
||||
estimatedComplexity: 'low' | 'medium' | 'high';
|
||||
modelUsed: string;
|
||||
planDurationMs: number;
|
||||
};
|
||||
}
|
||||
|
||||
export interface OrchestratorPhase {
|
||||
id: string; // "phase-1", "phase-2"
|
||||
name: string; // Human-readable name
|
||||
description: string;
|
||||
order: number;
|
||||
status: PhaseStatus;
|
||||
tasks: OrchestratorTask[];
|
||||
verificationCriteria: string[];
|
||||
testCommands: string[];
|
||||
maxAttempts: number; // Default: 3
|
||||
attempts: number; // Current attempt count
|
||||
startedAt: number | null;
|
||||
completedAt: number | null;
|
||||
durationMs: number | null;
|
||||
teamStrategy: TeamStrategy;
|
||||
}
|
||||
|
||||
export type PhaseStatus =
|
||||
| 'pending'
|
||||
| 'executing'
|
||||
| 'verifying'
|
||||
| 'passed'
|
||||
| 'failed'
|
||||
| 'skipped';
|
||||
|
||||
export interface OrchestratorTask {
|
||||
id: string; // "phase-1-task-1"
|
||||
phaseId: string;
|
||||
prompt: string; // Single-line prompt for Claude
|
||||
status: 'pending' | 'running' | 'completed' | 'failed';
|
||||
assignedSessionId: string | null;
|
||||
queueTaskId: string | null; // Links to TaskQueue task
|
||||
parallel: boolean; // Can run in parallel with sibling tasks
|
||||
completionPhrase: string; // Unique phrase for completion detection
|
||||
timeoutMs: number;
|
||||
startedAt: number | null;
|
||||
completedAt: number | null;
|
||||
error: string | null;
|
||||
retries: number;
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Team Strategy
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
export type TeamStrategy =
|
||||
| { type: 'single' } // One session handles all
|
||||
| { type: 'parallel'; maxSessions: number } // Multiple sessions
|
||||
| { type: 'team'; config: TeamSetup } // Agent teams
|
||||
|
||||
export interface TeamSetup {
|
||||
leadPrompt: string;
|
||||
suggestedTeammates: string[]; // Role descriptions
|
||||
maxTeammates: number;
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Verification
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
export interface VerificationResult {
|
||||
passed: boolean;
|
||||
checks: VerificationCheck[];
|
||||
summary: string;
|
||||
suggestions: string[]; // Recovery hints for replanning
|
||||
}
|
||||
|
||||
export interface VerificationCheck {
|
||||
type: 'test_command' | 'ai_review' | 'file_check';
|
||||
description: string;
|
||||
passed: boolean;
|
||||
output?: string;
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Configuration
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
export interface OrchestratorConfig {
|
||||
plannerModel: string; // Default: 'opus'
|
||||
researchEnabled: boolean; // Default: true
|
||||
autoApprove: boolean; // Default: false
|
||||
maxPhaseRetries: number; // Default: 3
|
||||
phaseTimeoutMs: number; // Default: 1800000 (30min)
|
||||
enableTeamAgents: boolean; // Default: true
|
||||
maxParallelSessions: number; // Default: 3
|
||||
verificationMode: 'strict' | 'moderate' | 'lenient';
|
||||
compactBetweenPhases: boolean; // Default: true
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Persistence (saved to ~/.codeman/state.json)
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
export interface OrchestratorPersistState {
|
||||
state: OrchestratorState;
|
||||
plan: OrchestratorPlan | null;
|
||||
currentPhaseIndex: number;
|
||||
startedAt: number | null;
|
||||
completedAt: number | null;
|
||||
config: OrchestratorConfig;
|
||||
stats: OrchestratorStats;
|
||||
}
|
||||
|
||||
export interface OrchestratorStats {
|
||||
phasesCompleted: number;
|
||||
phasesFailed: number;
|
||||
totalTasksCompleted: number;
|
||||
totalTasksFailed: number;
|
||||
totalDurationMs: number;
|
||||
replanCount: number;
|
||||
}
|
||||
```
|
||||
|
||||
## New Files (Implementation Order)
|
||||
|
||||
### Step 1: `src/types/orchestrator.ts` — Type definitions
|
||||
All interfaces above. No dependencies. ~120 lines.
|
||||
|
||||
### Step 2: `src/orchestrator-planner.ts` — Plan generation + phase grouping
|
||||
~300 lines. Wraps existing PlanOrchestrator.
|
||||
|
||||
```typescript
|
||||
/**
|
||||
* @fileoverview Orchestrator plan generation — converts goals into phased plans.
|
||||
*
|
||||
* Uses PlanOrchestrator for AI plan generation, then groups PlanItems into
|
||||
* sequential phases with team strategies and verification criteria.
|
||||
*
|
||||
* @module orchestrator-planner
|
||||
*/
|
||||
|
||||
export class OrchestratorPlanner {
|
||||
constructor(mux: TerminalMultiplexer, workingDir: string, config: OrchestratorConfig);
|
||||
|
||||
/** Generate plan from goal. Uses PlanOrchestrator internally. */
|
||||
async generatePlan(goal: string, onProgress?: ProgressCallback): Promise<OrchestratorPlan>;
|
||||
|
||||
/** Cancel in-progress plan generation. */
|
||||
async cancel(): Promise<void>;
|
||||
|
||||
// Internal
|
||||
private groupIntoPhases(items: PlanItem[], goal: string): OrchestratorPhase[];
|
||||
private assignTeamStrategies(phases: OrchestratorPhase[]): void;
|
||||
private generateCompletionPhrases(plan: OrchestratorPlan): void;
|
||||
}
|
||||
```
|
||||
|
||||
**Phase grouping algorithm:**
|
||||
1. Topological sort by `PlanItem.dependencies`
|
||||
2. Group into dependency layers (Kahn's algorithm)
|
||||
3. Within each layer, sub-group by `tddPhase` (setup → test → impl → verify → review)
|
||||
4. Merge adjacent small phases (< 2 tasks) if they share the same tddPhase
|
||||
5. Assign team strategies:
|
||||
- 1-2 tasks → `{ type: 'single' }`
|
||||
- 3+ independent tasks → `{ type: 'parallel', maxSessions: Math.min(taskCount, config.maxParallelSessions) }`
|
||||
- 4+ tasks with high complexity → `{ type: 'team', config: { ... } }`
|
||||
6. Generate unique completion phrases per task: `ORCH_P{phaseOrder}_T{taskIndex}`
|
||||
|
||||
### Step 3: `src/orchestrator-verifier.ts` — Phase verification
|
||||
~200 lines.
|
||||
|
||||
```typescript
|
||||
/**
|
||||
* @fileoverview Orchestrator phase verification.
|
||||
*
|
||||
* Runs verification checks after each phase completes:
|
||||
* test commands, AI review, and file existence checks.
|
||||
*
|
||||
* @module orchestrator-verifier
|
||||
*/
|
||||
|
||||
export class OrchestratorVerifier {
|
||||
constructor(config: OrchestratorConfig);
|
||||
|
||||
/** Run all verification checks for a completed phase. */
|
||||
async verifyPhase(
|
||||
phase: OrchestratorPhase,
|
||||
session: Session,
|
||||
mode: 'strict' | 'moderate' | 'lenient'
|
||||
): Promise<VerificationResult>;
|
||||
|
||||
// Verification strategies
|
||||
private async runTestCommands(commands: string[], session: Session): Promise<VerificationCheck[]>;
|
||||
private async aiReview(phase: OrchestratorPhase, session: Session): Promise<VerificationCheck>;
|
||||
}
|
||||
```
|
||||
|
||||
**Verification modes:**
|
||||
- `strict`: ALL test commands must pass AND AI review must approve
|
||||
- `moderate`: Test commands must pass, AI review is advisory
|
||||
- `lenient`: At least one test command passes, AI review skipped
|
||||
|
||||
**AI review prompt (sent as a task to the session):**
|
||||
```
|
||||
Review Phase "{phase.name}" completion. Check:
|
||||
1. Expected functionality works
|
||||
2. No obvious regressions
|
||||
3. Code quality is acceptable
|
||||
|
||||
Criteria: {phase.verificationCriteria.join('\n')}
|
||||
|
||||
If ALL criteria are met, respond: ORCH_VERIFY_PASS
|
||||
If ANY criteria fail, respond: ORCH_VERIFY_FAIL and explain what failed.
|
||||
```
|
||||
|
||||
### Step 4: `src/orchestrator-loop.ts` — Core state machine
|
||||
~500 lines. Main orchestrator engine.
|
||||
|
||||
```typescript
|
||||
/**
|
||||
* @fileoverview Orchestrator Loop — phased plan execution with team agents.
|
||||
*
|
||||
* State machine that generates plans from user goals, executes them
|
||||
* phase-by-phase with verification gates, and adapts on failure.
|
||||
*
|
||||
* @module orchestrator-loop
|
||||
*/
|
||||
|
||||
export interface OrchestratorLoopEvents {
|
||||
stateChanged: (state: OrchestratorState, prevState: OrchestratorState) => void;
|
||||
planReady: (plan: OrchestratorPlan) => void;
|
||||
phaseStarted: (phase: OrchestratorPhase) => void;
|
||||
phaseCompleted: (phase: OrchestratorPhase) => void;
|
||||
phaseFailed: (phase: OrchestratorPhase, reason: string) => void;
|
||||
taskAssigned: (task: OrchestratorTask, sessionId: string) => void;
|
||||
taskCompleted: (task: OrchestratorTask) => void;
|
||||
taskFailed: (task: OrchestratorTask, error: string) => void;
|
||||
verificationResult: (phase: OrchestratorPhase, result: VerificationResult) => void;
|
||||
completed: (stats: OrchestratorStats) => void;
|
||||
error: (error: Error) => void;
|
||||
}
|
||||
|
||||
export class OrchestratorLoop extends EventEmitter {
|
||||
private state: OrchestratorState = 'idle';
|
||||
private plan: OrchestratorPlan | null = null;
|
||||
private currentPhaseIndex = 0;
|
||||
private config: OrchestratorConfig;
|
||||
private planner: OrchestratorPlanner;
|
||||
private verifier: OrchestratorVerifier;
|
||||
private sessionManager: SessionManager;
|
||||
private taskQueue: TaskQueue;
|
||||
private store: StateStore;
|
||||
private stats: OrchestratorStats;
|
||||
private cleanup: CleanupManager;
|
||||
private pausedState: OrchestratorState | null = null; // State before pause
|
||||
|
||||
// ── Lifecycle ──────────────────────────────────────────────
|
||||
|
||||
constructor(mux: TerminalMultiplexer, workingDir: string, config?: Partial<OrchestratorConfig>);
|
||||
|
||||
/** Start orchestration with a goal. Transitions: idle → planning */
|
||||
async start(goal: string): Promise<void>;
|
||||
|
||||
/** Approve the generated plan. Transitions: approval → executing */
|
||||
async approve(): Promise<void>;
|
||||
|
||||
/** Reject plan with feedback. Transitions: approval → planning (regenerate) */
|
||||
async reject(feedback: string): Promise<void>;
|
||||
|
||||
/** Pause execution. Saves current state. */
|
||||
pause(): void;
|
||||
|
||||
/** Resume from pause. */
|
||||
resume(): void;
|
||||
|
||||
/** Stop everything and clean up. → idle */
|
||||
async stop(): Promise<void>;
|
||||
|
||||
/** Skip current phase. → executing (next phase) or completed */
|
||||
async skipPhase(phaseId: string): Promise<void>;
|
||||
|
||||
/** Retry a failed phase. → executing */
|
||||
async retryPhase(phaseId: string): Promise<void>;
|
||||
|
||||
// ── Getters ────────────────────────────────────────────────
|
||||
|
||||
getState(): OrchestratorState;
|
||||
getPlan(): OrchestratorPlan | null;
|
||||
getCurrentPhase(): OrchestratorPhase | null;
|
||||
getStats(): OrchestratorStats;
|
||||
getStatus(): OrchestratorPersistState;
|
||||
|
||||
// ── Internal: Phase Execution ──────────────────────────────
|
||||
|
||||
private async executeCurrentPhase(): Promise<void>;
|
||||
private async executePhase(phase: OrchestratorPhase): Promise<void>;
|
||||
private async assignPhaseTasks(phase: OrchestratorPhase): Promise<void>;
|
||||
private handleTaskCompleted(taskId: string): void;
|
||||
private handleTaskFailed(taskId: string, error: string): void;
|
||||
private async onPhaseTasksComplete(phase: OrchestratorPhase): Promise<void>;
|
||||
|
||||
// ── Internal: Verification ─────────────────────────────────
|
||||
|
||||
private async verifyCurrentPhase(): Promise<void>;
|
||||
private async handleVerificationResult(phase: OrchestratorPhase, result: VerificationResult): Promise<void>;
|
||||
|
||||
// ── Internal: Replanning ───────────────────────────────────
|
||||
|
||||
private async replanPhase(phase: OrchestratorPhase, failures: string[]): Promise<void>;
|
||||
|
||||
// ── Internal: State Machine ────────────────────────────────
|
||||
|
||||
private setState(newState: OrchestratorState): void;
|
||||
private advanceToNextPhase(): Promise<void>;
|
||||
private persist(): void;
|
||||
private restore(): void;
|
||||
}
|
||||
```
|
||||
|
||||
**Key execution flow in `executePhase()`:**
|
||||
1. Mark phase as `executing`, emit `phaseStarted`
|
||||
2. For each task in phase:
|
||||
- Create a `CreateTaskOptions` from `OrchestratorTask`
|
||||
- Add to `TaskQueue` with proper dependencies + completion phrase
|
||||
- Store the TaskQueue task ID in `OrchestratorTask.queueTaskId`
|
||||
3. Poll task completion (listen to TaskQueue events)
|
||||
4. When all tasks complete → call `onPhaseTasksComplete()`
|
||||
5. `onPhaseTasksComplete()` triggers verification
|
||||
|
||||
**How tasks get assigned to sessions:**
|
||||
The OrchestratorLoop does NOT manage session assignment directly. It adds tasks to the existing TaskQueue and starts a mini poll loop that assigns pending tasks to idle sessions — the same pattern as RalphLoop's `assignTasks()`. This reuses existing session management.
|
||||
|
||||
**Team agent flow:**
|
||||
For phases with `teamStrategy.type === 'team'`:
|
||||
- Start a single session with `CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1`
|
||||
- Instead of adding individual tasks to TaskQueue, send ONE comprehensive prompt to the lead
|
||||
- The prompt instructs the lead to create teammates and delegate
|
||||
- Monitor via TeamWatcher for team task completion + hook events
|
||||
- Phase completion is detected via the lead's completion phrase
|
||||
|
||||
### Step 5: `src/web/routes/orchestrator-routes.ts` — API endpoints
|
||||
~300 lines.
|
||||
|
||||
```
|
||||
POST /api/orchestrator/start — { goal, config? } → start planning
|
||||
POST /api/orchestrator/approve — approve generated plan
|
||||
POST /api/orchestrator/reject — { feedback } → reject + replan
|
||||
POST /api/orchestrator/pause — pause execution
|
||||
POST /api/orchestrator/resume — resume execution
|
||||
POST /api/orchestrator/stop — stop orchestration
|
||||
GET /api/orchestrator/status — full state + plan + stats
|
||||
GET /api/orchestrator/plan — plan details only
|
||||
POST /api/orchestrator/phase/:id/skip — skip a phase
|
||||
POST /api/orchestrator/phase/:id/retry — retry a failed phase
|
||||
```
|
||||
|
||||
Port dependency: `SessionPort & EventPort & RespawnPort & ConfigPort & InfraPort`
|
||||
|
||||
The route module receives the OrchestratorLoop instance via the InfraPort (added to `createRouteContext()`).
|
||||
|
||||
### Step 6: SSE Events — `src/web/sse-events.ts` additions
|
||||
|
||||
```typescript
|
||||
// ─── Orchestrator ────────────────────────────────────────────────────────────
|
||||
|
||||
/** Orchestrator state machine transitioned. */
|
||||
export const OrchestratorStateChanged = 'orchestrator:stateChanged' as const;
|
||||
/** Orchestrator plan generated and ready for approval. */
|
||||
export const OrchestratorPlanReady = 'orchestrator:planReady' as const;
|
||||
/** Orchestrator phase started executing. */
|
||||
export const OrchestratorPhaseStarted = 'orchestrator:phaseStarted' as const;
|
||||
/** Orchestrator phase completed successfully. */
|
||||
export const OrchestratorPhaseCompleted = 'orchestrator:phaseCompleted' as const;
|
||||
/** Orchestrator phase failed. */
|
||||
export const OrchestratorPhaseFailed = 'orchestrator:phaseFailed' as const;
|
||||
/** Orchestrator verification result for a phase. */
|
||||
export const OrchestratorVerification = 'orchestrator:verification' as const;
|
||||
/** Orchestrator task assigned to session. */
|
||||
export const OrchestratorTaskAssigned = 'orchestrator:taskAssigned' as const;
|
||||
/** Orchestrator task completed. */
|
||||
export const OrchestratorTaskCompleted = 'orchestrator:taskCompleted' as const;
|
||||
/** Orchestrator task failed. */
|
||||
export const OrchestratorTaskFailed = 'orchestrator:taskFailed' as const;
|
||||
/** All phases completed successfully. */
|
||||
export const OrchestratorCompleted = 'orchestrator:completed' as const;
|
||||
/** Orchestrator error. */
|
||||
export const OrchestratorError = 'orchestrator:error' as const;
|
||||
```
|
||||
|
||||
11 new events. Add to `SseEvent` namespace object + mirror in `constants.js`.
|
||||
|
||||
### Step 7: State persistence — `src/state-store.ts` additions
|
||||
|
||||
Add to `AppState`:
|
||||
```typescript
|
||||
orchestrator?: OrchestratorPersistState;
|
||||
```
|
||||
|
||||
Add methods:
|
||||
```typescript
|
||||
getOrchestratorState(): OrchestratorPersistState | null;
|
||||
setOrchestratorState(state: Partial<OrchestratorPersistState>): void;
|
||||
clearOrchestratorState(): void;
|
||||
```
|
||||
|
||||
### Step 8: Server integration — `src/web/server.ts` modifications
|
||||
|
||||
1. Import `OrchestratorLoop` and `registerOrchestratorRoutes`
|
||||
2. Add `private orchestratorLoop: OrchestratorLoop` field
|
||||
3. Initialize in constructor (lazy — created on first start, not at boot)
|
||||
4. Add to `createRouteContext()` InfraPort: `orchestratorLoop: this.orchestratorLoop`
|
||||
5. Wire up OrchestratorLoop events → SSE broadcasts
|
||||
6. Register routes: `registerOrchestratorRoutes(this.app, ctx)`
|
||||
7. Clean up in `stop()`
|
||||
|
||||
### Step 9: `src/web/public/orchestrator-ui.js` — Frontend panel
|
||||
~500 lines. New frontend module.
|
||||
|
||||
**Load order**: After `panels-ui.js` (11), before `ralph-wizard.js` (13). So load order = 11.5.
|
||||
|
||||
**UI elements:**
|
||||
- Goal input form (text area + config toggles)
|
||||
- Plan approval view (phase list, task details, approve/reject buttons)
|
||||
- Execution dashboard (progress bar, phase cards, task status indicators)
|
||||
- Agent activity panel (session count, team status)
|
||||
- Controls (pause, resume, stop, skip phase, retry phase)
|
||||
|
||||
**SSE listeners:**
|
||||
- All 11 orchestrator events → update UI state
|
||||
- Reuses existing session/respawn/team event handlers for agent monitoring
|
||||
|
||||
### Step 10: `src/prompts/orchestrator.ts` — Prompt templates
|
||||
~200 lines.
|
||||
|
||||
Templates for:
|
||||
- Phase execution prompt (tells Claude what to do in this phase)
|
||||
- Team lead delegation prompt (instructs lead to create and coordinate teammates)
|
||||
- Verification prompt (asks Claude to verify phase output)
|
||||
- Replan prompt (gives failure context, asks for recovery steps)
|
||||
|
||||
### Step 11: Constants, schemas, route barrel updates
|
||||
|
||||
- `src/web/public/constants.js` — Add 11 SSE event mirrors
|
||||
- `src/web/schemas.ts` — Add Zod schemas for orchestrator API input validation
|
||||
- `src/web/routes/index.ts` — Export `registerOrchestratorRoutes`
|
||||
- `src/web/ports/infra-port.ts` — Add `orchestratorLoop` to InfraPort
|
||||
- `src/types/index.ts` — Export orchestrator types
|
||||
|
||||
## Existing File Modifications Summary
|
||||
|
||||
| File | Change | Lines |
|
||||
|------|--------|-------|
|
||||
| `src/types/index.ts` | Add orchestrator barrel export | +1 |
|
||||
| `src/web/sse-events.ts` | Add 11 orchestrator events + SseEvent entries | +30 |
|
||||
| `src/web/public/constants.js` | Mirror 11 SSE events | +15 |
|
||||
| `src/web/routes/index.ts` | Export registerOrchestratorRoutes | +1 |
|
||||
| `src/web/ports/infra-port.ts` | Add orchestratorLoop to InfraPort | +3 |
|
||||
| `src/web/server.ts` | Initialize OrchestratorLoop, wire events, register routes | +40 |
|
||||
| `src/web/schemas.ts` | Add orchestrator Zod schemas | +20 |
|
||||
| `src/state-store.ts` | Add orchestrator state persistence | +20 |
|
||||
| `src/web/public/app.js` | Add orchestrator SSE listeners + panel toggle | +30 |
|
||||
| `src/web/public/index.html` | Add orchestrator-ui.js script tag | +1 |
|
||||
|
||||
**Total new code**: ~2,300 lines across 6 new files
|
||||
**Total modifications**: ~160 lines across 10 existing files
|
||||
|
||||
## Implementation Execution Order
|
||||
|
||||
This is the actual build order — each step is a commit checkpoint:
|
||||
|
||||
1. **Types** — `src/types/orchestrator.ts` + barrel export. Zero risk, pure types.
|
||||
2. **SSE events** — Add all 11 events to both `sse-events.ts` and `constants.js`. Wire in SseEvent namespace.
|
||||
3. **State persistence** — Add orchestrator state to StateStore. Small, isolated change.
|
||||
4. **Schemas** — Add Zod validation schemas for API input.
|
||||
5. **Planner** — `src/orchestrator-planner.ts`. Can test in isolation.
|
||||
6. **Verifier** — `src/orchestrator-verifier.ts`. Can test in isolation.
|
||||
7. **Core loop** — `src/orchestrator-loop.ts`. The big one. Depends on planner + verifier.
|
||||
8. **Prompts** — `src/prompts/orchestrator.ts`. Templates used by core loop.
|
||||
9. **Port + routes** — `src/web/ports/infra-port.ts` update + `src/web/routes/orchestrator-routes.ts`.
|
||||
10. **Server integration** — Wire OrchestratorLoop into WebServer. Routes become live.
|
||||
11. **Frontend** — `src/web/public/orchestrator-ui.js` + app.js listeners + index.html script tag.
|
||||
12. **Tests** — `test/orchestrator-*.test.ts`.
|
||||
13. **Typecheck + lint** — Fix all issues, ensure CI passes.
|
||||
|
||||
## Edge Cases & Error Handling
|
||||
|
||||
- **Session limit reached**: Queue tasks and wait for sessions to free up (existing SessionManager handles this)
|
||||
- **All sessions crash during phase**: Mark phase as failed, attempt replan
|
||||
- **Verification flaky**: `moderate` mode allows test retries; `lenient` skips AI review
|
||||
- **Plan too large**: Cap at 10 phases, 50 total tasks. Warn user.
|
||||
- **Context overflow**: Auto-compact between phases. Respawn if needed (orchestrator state is external).
|
||||
- **User pauses mid-phase**: Pause task assignment, don't cancel running tasks. Resume picks up where it left off.
|
||||
- **Network/API errors during planning**: Retry plan generation up to 2 times, then fail with clear message.
|
||||
- **Orchestrator vs Ralph conflict**: Mutually exclusive. Starting orchestrator stops Ralph if running. Starting Ralph stops orchestrator.
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
- **Unit tests**: `test/orchestrator-planner.test.ts` — phase grouping algorithm, team strategy assignment
|
||||
- **Unit tests**: `test/orchestrator-verifier.test.ts` — verification logic with mocked sessions
|
||||
- **Integration tests**: `test/orchestrator-loop.test.ts` — state machine transitions, task lifecycle
|
||||
- **Route tests**: `test/routes/orchestrator-routes.test.ts` — API validation, status responses
|
||||
|
||||
All tests use `MockSession` pattern from existing test infrastructure. No real tmux needed.
|
||||
@@ -0,0 +1,157 @@
|
||||
# Orchestrator Loop — Research Findings
|
||||
|
||||
> Research doc for the new "Orchestrator Loop" feature. Not for GitHub.
|
||||
|
||||
## What We're Building
|
||||
|
||||
A new autonomous loop variant — **Orchestrator Loop** — that takes high-level user tasks, decomposes them into a detailed plan using team agents, and executes the plan step-by-step with quality gates. Unlike Ralph Loop (which executes a flat task queue), the Orchestrator coordinates **planning, delegation, and verification** as a continuous cycle.
|
||||
|
||||
**Core idea**: User inputs a goal → Orchestrator creates a detailed plan → spins up team agents for parallel execution → validates each step → adapts the plan based on results → delivers polished output.
|
||||
|
||||
## Existing Infrastructure Analysis
|
||||
|
||||
### What We Can Reuse
|
||||
|
||||
#### 1. Ralph Loop (`src/ralph-loop.ts`)
|
||||
- **Pattern**: Poll loop with `start() → tick() → stop()` lifecycle
|
||||
- **Reusable**: Event-driven task assignment, session completion handling, timeout management
|
||||
- **Limitation**: Flat task queue — no concept of phases, dependencies between task groups, or adaptive replanning
|
||||
- **Key insight**: `assignTaskToSession()` uses `session.sendInput(task.prompt)` — simple prompt injection into PTY
|
||||
|
||||
#### 2. Task Queue (`src/task-queue.ts`) + Task (`src/task.ts`)
|
||||
- **Already has**: Priority ordering, dependency tracking between tasks, completion phrase detection
|
||||
- **Limitation**: No task *groups* or *phases*. Dependencies are task-to-task, not phase-to-phase
|
||||
- **Key insight**: Tasks support `completionPhrase` — a string the task watches for in output. This is how Ralph knows a task is done
|
||||
|
||||
#### 3. Plan Orchestrator (`src/plan-orchestrator.ts`)
|
||||
- **Already has**: 2-agent plan generation (Research Agent → Planner Agent), TDD-aware plan items with P0/P1/P2 priorities
|
||||
- **Output**: `PlanItem[]` with dependencies, verification criteria, TDD phases, complexity ratings
|
||||
- **Limitation**: Plan generation only — no execution. Plans are generated then sit in state/UI for human review
|
||||
- **Key insight**: Uses `Session` directly to run Claude subagent instances for research and planning. Returns structured JSON
|
||||
|
||||
#### 4. Team Agents (`src/team-watcher.ts`, `~/.claude/teams/`)
|
||||
- **Already has**: Team creation, member tracking, filesystem inbox messaging, task management via `~/.claude/tasks/{team-name}/`
|
||||
- **Limitation**: Codeman can only *observe* teams (TeamWatcher is read-only polling), not *create* or *orchestrate* them
|
||||
- **Key insight**: Teams are a Claude Code feature. Codeman monitors them but doesn't control them. We can't programmatically create teammates — Claude Code does that when you use `CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1`
|
||||
|
||||
#### 5. Respawn Controller (`src/respawn-controller.ts`)
|
||||
- **Already has**: Preset-based automation (ralph-todo, overnight-autonomous), circuit breaker, health scoring
|
||||
- **Key insight**: The `ralph-todo` preset (8s idle, 480min max) is designed for autonomous task execution. We'd need a new preset or make Orchestrator Loop set its own timing
|
||||
|
||||
#### 6. Session Auto-Ops (`src/session-auto-ops.ts`)
|
||||
- **Already has**: Auto-compact at token thresholds, auto-clear for context management
|
||||
- **Key insight**: Critical for long Orchestrator runs — prevents context overflow during multi-step execution
|
||||
|
||||
#### 7. Hooks (`src/hooks-config.ts`)
|
||||
- **Already has**: `idle_prompt`, `stop`, `teammate_idle`, `task_completed` hook events
|
||||
- **Key insight**: Hooks fire POST to `/api/hook-event` — this is how Codeman knows when Claude is idle, stopped, or completed a task. The Orchestrator Loop can listen to these same events
|
||||
|
||||
### What We Need to Build New
|
||||
|
||||
1. **Plan → Task decomposition**: Convert PlanOrchestrator output (PlanItem[]) into executable task groups with phase ordering
|
||||
2. **Multi-phase execution engine**: Execute plan phases sequentially, tasks within phases in parallel
|
||||
3. **Verification gates**: After each phase, run verification (test commands, AI review) before proceeding
|
||||
4. **Adaptive replanning**: When a task fails or verification fails, generate a recovery plan
|
||||
5. **Team agent orchestration**: Leverage Claude Code's agent teams for parallel execution within phases
|
||||
6. **Progress tracking & UI**: Real-time dashboard showing plan progress, phase status, agent activity
|
||||
|
||||
## How Teams Actually Work (Important Constraint)
|
||||
|
||||
After deep research, here's the reality of agent teams:
|
||||
|
||||
```
|
||||
User starts session with CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1
|
||||
→ Claude Code creates a team-lead
|
||||
→ Team-lead spawns teammates (in-process threads)
|
||||
→ Teammates appear as subagents (detected by SubagentWatcher)
|
||||
→ Communication via ~/.claude/teams/{name}/inboxes/{member}.json
|
||||
→ Tasks tracked in ~/.claude/tasks/{team-name}/{N}.json
|
||||
```
|
||||
|
||||
**Codeman cannot programmatically create team members.** This is a Claude Code internal feature. However, Codeman CAN:
|
||||
- Start a session that has teams enabled
|
||||
- Send a prompt to the lead that instructs it to use agent teams
|
||||
- Monitor team activity via TeamWatcher
|
||||
- React to teammate_idle and task_completed hook events
|
||||
- Read team task status from the filesystem
|
||||
|
||||
**This means**: The Orchestrator Loop orchestrates at the *session prompt* level, not the *team member* level. We tell the lead what to do, and the lead decides how to use its team.
|
||||
|
||||
## Architecture Decision: Prompt-Level Orchestration
|
||||
|
||||
Given the team constraint, the Orchestrator Loop works by:
|
||||
|
||||
1. **Planning phase**: Use PlanOrchestrator to generate a detailed plan from user input
|
||||
2. **Execution phase**: Feed plan steps as prompts to sessions, one phase at a time
|
||||
3. **Verification phase**: After each phase, run verification prompts and check results
|
||||
4. **Adaptation phase**: If verification fails, generate recovery prompts
|
||||
|
||||
The "team agents" aspect works by:
|
||||
- Starting sessions with `CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1`
|
||||
- Crafting prompts that *instruct the lead to delegate* to teammates
|
||||
- Monitoring team activity to track parallel progress
|
||||
- The lead agent is smart enough to decompose work across its team
|
||||
|
||||
## Key Technical Findings
|
||||
|
||||
### Session Input Mechanics
|
||||
```typescript
|
||||
// From session.ts - how we send prompts
|
||||
await session.sendInput(task.prompt); // Uses writeViaMux() internally
|
||||
// writeViaMux() does: tmux send-keys -l "prompt text" + tmux send-keys Enter
|
||||
// CRITICAL: Single-line only! Multi-line breaks Ink rendering
|
||||
```
|
||||
|
||||
### Completion Detection Chain
|
||||
```
|
||||
PTY output → RalphTracker.processData() → completion phrase fuzzy match
|
||||
→ CompletionConfidence scoring (multi-signal: promise tag + todos + exit signal)
|
||||
→ If confident → emit 'completionDetected'
|
||||
→ RalphLoop listens → marks task complete → assigns next
|
||||
```
|
||||
|
||||
### How Plan Items Map to Tasks
|
||||
```typescript
|
||||
// PlanItem has:
|
||||
interface PlanItem {
|
||||
id: string; // "P0-001"
|
||||
content: string; // "Implement error handling for API endpoints"
|
||||
priority: 'P0' | 'P1' | 'P2';
|
||||
dependencies: string[]; // ["P0-000"] — other PlanItem IDs
|
||||
verificationCriteria: string;
|
||||
testCommand: string;
|
||||
tddPhase: 'setup' | 'test' | 'impl' | 'verify' | 'review';
|
||||
complexity: 'low' | 'medium' | 'high';
|
||||
}
|
||||
|
||||
// Task has:
|
||||
interface CreateTaskOptions {
|
||||
prompt: string;
|
||||
priority: number;
|
||||
dependencies: string[]; // Task IDs
|
||||
completionPhrase: string;
|
||||
timeoutMs: number;
|
||||
}
|
||||
|
||||
// Natural mapping: PlanItem.content → Task.prompt
|
||||
// PlanItem.dependencies → Task.dependencies
|
||||
// PlanItem.priority → Task.priority (P0=100, P1=50, P2=10)
|
||||
// PlanItem.verificationCriteria → verification task prompt
|
||||
```
|
||||
|
||||
### Context Management for Long Runs
|
||||
- Auto-compact at ~110k tokens (configurable)
|
||||
- Auto-clear at ~140k tokens (configurable)
|
||||
- Respawn cycling: kill + restart session to reset context entirely
|
||||
- For Orchestrator: we want compact between phases, respawn between major milestones
|
||||
|
||||
## Risk Assessment
|
||||
|
||||
| Risk | Severity | Mitigation |
|
||||
|------|----------|------------|
|
||||
| Context overflow during complex phases | High | Auto-compact between tasks, respawn between phases |
|
||||
| Team agents not predictable | Medium | Orchestrate at session level, let Claude decide team delegation |
|
||||
| Plan too ambitious → infinite loop | High | Phase budgets (max attempts per phase), circuit breaker |
|
||||
| Verification too strict → blocks progress | Medium | Configurable strictness, human override via UI |
|
||||
| Single-line prompt limit | Medium | Use CLAUDE.md file for complex instructions, prompt references file |
|
||||
| Long planning phase delays execution | Low | Show plan for approval before execution |
|
||||
@@ -0,0 +1,319 @@
|
||||
# Security Architecture
|
||||
|
||||
This document describes Codeman's security model: how it decides who may reach
|
||||
the web UI, how requests are authenticated, how the file-serving and tmux layers
|
||||
are hardened, and the recommended ways to expose an instance safely.
|
||||
|
||||
Codeman spawns and drives Claude/OpenCode CLIs with
|
||||
`--dangerously-skip-permissions`. **Anyone who can reach an unauthenticated
|
||||
instance can run arbitrary commands as your user.** The defaults below are chosen
|
||||
so that a fresh install is safe on the machine it runs on, while remote access is
|
||||
an explicit, guided opt‑in.
|
||||
|
||||
> TL;DR — Codeman binds **loopback only (`127.0.0.1`) by default**, so out of the
|
||||
> box it is reachable only from the same machine and needs no password. To reach
|
||||
> it from elsewhere, either put it behind an **authenticated tunnel**
|
||||
> (`tailscale serve` / `cloudflared`) **or** bind a wider host **and set
|
||||
> `CODEMAN_PASSWORD`**. If you bind a non‑loopback host with no password, Codeman
|
||||
> still starts but prints a **loud warning** telling you how to secure it.
|
||||
|
||||
---
|
||||
|
||||
## 1. Network binding model
|
||||
|
||||
| Setting | Default | Source |
|
||||
|---------|---------|--------|
|
||||
| Bind host | `127.0.0.1` (loopback) | `--host` / `CODEMAN_HOST` → `WebServer` ctor |
|
||||
| Port | `3000` | `--port` / `CODEMAN_PORT` |
|
||||
| TLS | off (`--https` to enable) | `--https` |
|
||||
|
||||
### Bind host classification
|
||||
|
||||
`isLoopbackBindHost()` (`src/web/network-auth-policy.ts`) decides whether a bind
|
||||
host is loopback-only. It returns `true` for:
|
||||
|
||||
- `localhost`
|
||||
- any IPv4 in `127.0.0.0/8` (e.g. `127.0.0.1`, `127.42.0.9`)
|
||||
- IPv6 loopback `::1` (bracketed `[::1]` and the long form `0:0:0:0:0:0:0:1`)
|
||||
- IPv4‑mapped loopback `::ffff:127.*`
|
||||
|
||||
It returns `false` for `0.0.0.0`, `::` (all interfaces), LAN IPs, and hostnames.
|
||||
The classification is **fail‑safe in the dangerous direction**: any host that is
|
||||
not provably loopback is treated as non‑loopback (it never mistakes `0.0.0.0`
|
||||
for loopback). Shorthand forms like `127.1` or integer/octal IPs classify as
|
||||
non‑loopback (you'll get a warning, not a silent wide‑open bind) — use
|
||||
`127.0.0.1` for an unambiguous loopback bind.
|
||||
|
||||
### Startup policy (the "warn, don't block" rule)
|
||||
|
||||
At `WebServer.start()`:
|
||||
|
||||
| Bind host | `CODEMAN_PASSWORD` | Behavior |
|
||||
|-----------|--------------------|----------|
|
||||
| loopback (default) | unset | **Start.** Safe — reachable only from this machine. |
|
||||
| loopback | set | **Start.** Auth required even locally. |
|
||||
| non‑loopback | set | **Start.** Auth protects the open bind. |
|
||||
| non‑loopback | unset | **Start + LOUD warning** listing how to secure it. |
|
||||
| non‑loopback | unset, `--allow-unauthenticated-network` | **Start + terse acknowledged note.** |
|
||||
|
||||
> History: an earlier iteration (unreleased COD‑29) *refused to start* on a
|
||||
> non‑loopback bind without a password. That surprised setups that "just worked"
|
||||
> before, so **0.9.0 changed it to start‑and‑warn**. Loopback is still the safe
|
||||
> default; the warning (with three concrete fixes) replaces the hard failure.
|
||||
|
||||
The warning points at three ways to secure the instance:
|
||||
|
||||
1. `CODEMAN_PASSWORD=<password>` — turns on HTTP Basic auth (see §2).
|
||||
2. `--host 127.0.0.1` + an authenticated tunnel (`cloudflared` / `tailscale serve`).
|
||||
3. `--allow-unauthenticated-network` / `CODEMAN_ALLOW_UNAUTHENTICATED_NETWORK=1`
|
||||
— explicitly accept the risk (downgrades the warning to a one‑line note). This
|
||||
flag is **only** an acknowledgement; it does not change reachability.
|
||||
|
||||
`CODEMAN_API_URL` (used by hooks/child processes) is always derived as a loopback
|
||||
address (`0.0.0.0`/`localhost`/`::1` → `127.0.0.1`) so in‑process hooks reach the
|
||||
server over loopback regardless of the public bind.
|
||||
|
||||
---
|
||||
|
||||
## 2. Authentication
|
||||
|
||||
Auth is **optional** and controlled by env vars captured at startup:
|
||||
|
||||
- `CODEMAN_USERNAME` (default `admin` when only a password is set)
|
||||
- `CODEMAN_PASSWORD`
|
||||
|
||||
When `CODEMAN_PASSWORD` is unset, no auth is enforced — which is why the default
|
||||
loopback bind matters. The auth pipeline (`src/web/middleware/auth.ts`,
|
||||
`onRequest` hook) runs in this order:
|
||||
|
||||
1. **Localhost‑only exemptions** (always first): `POST /api/hook-event` and the QR
|
||||
`/q/` short‑code path are exempt when `req.ip` is loopback (see §3).
|
||||
2. **Session cookie** check — a valid `codeman_session` cookie short‑circuits to
|
||||
allow.
|
||||
3. **HTTP Basic** check — correct credentials short‑circuit to allow and clear
|
||||
that IP's failure counter.
|
||||
4. **Rate‑limit gate** — if neither cookie nor credentials passed and the IP is
|
||||
locked out, return `429` with a `Retry-After` header.
|
||||
5. Otherwise return `401`, incrementing the IP's failure counter.
|
||||
|
||||
### Session cookies
|
||||
|
||||
On successful Basic auth the server issues `codeman_session`, an opaque
|
||||
server‑side token (`randomBytes(32)`), valid 24h with auto‑extend and device
|
||||
context for the audit log. Tokens are **not** client‑signed — they're validated
|
||||
by presence in a server‑side map, so they cannot be forged offline.
|
||||
|
||||
### Rate limiting / lockout recovery
|
||||
|
||||
Failed auth is tracked **per IP**: 10 failures → `429`, with a 15‑minute decay.
|
||||
The QR path has its own separate limiter.
|
||||
|
||||
The lockout check sits **after** the cookie/credential checks (step 4, not first).
|
||||
This is deliberate: a user with a **valid cookie or correct password recovers
|
||||
immediately** even while an attacker is hammering the same IP — important because
|
||||
all traffic through a tunnel shares one source IP (loopback). Wrong credentials
|
||||
are still counted and still hit the `429` at the threshold, so brute‑force
|
||||
protection is unchanged.
|
||||
|
||||
---
|
||||
|
||||
## 3. Request‑origin trust & the tunnel caveat
|
||||
|
||||
`req.ip` is derived from the **TCP socket only** — Fastify runs with
|
||||
`trustProxy: false`, so `X-Forwarded-For` / `X-Real-IP` / `Forwarded` are
|
||||
**ignored**. A remote client cannot forge `req.ip` to `127.0.0.1`.
|
||||
|
||||
**However**, a reverse tunnel that connects to the server over loopback (e.g.
|
||||
`cloudflared --url http://localhost:3000`) makes **every tunneled request arrive
|
||||
with `req.ip = 127.0.0.1`**. The localhost‑only exemptions then treat those
|
||||
requests as local:
|
||||
|
||||
- `POST /api/hook-event` — auth‑exempt for loopback. Bounded impact: it is
|
||||
`HookEventSchema`‑validated and requires a valid in‑memory `sessionId`; it can
|
||||
drive respawn signals, SSE broadcasts, push notifications, and transcript
|
||||
watching — **not** arbitrary terminal input or file reads. It is a
|
||||
session‑disruption / notification‑spoofing surface, not RCE.
|
||||
- QR `/q/` — still protected by its own short‑code brute‑force limiter
|
||||
(10 failures / 60s against a 62⁶ space).
|
||||
|
||||
**Mitigation:** set `CODEMAN_PASSWORD` whenever a loopback‑connecting tunnel is
|
||||
up (it does not gate the hook‑event exemption, but it gates everything else and
|
||||
is the documented practice). Prefer `tailscale serve` (below), which authenticates
|
||||
at the tailnet layer so untrusted clients never reach the loopback port at all.
|
||||
A future hardening could gate the hook‑event exemption on a shared secret while a
|
||||
tunnel is active.
|
||||
|
||||
---
|
||||
|
||||
## 4. Recommended remote‑access setups
|
||||
|
||||
Ordered most‑to‑least recommended:
|
||||
|
||||
### A. Tailscale serve (recommended)
|
||||
|
||||
Bind loopback, let Tailscale front it on your tailnet with a real cert:
|
||||
|
||||
```bash
|
||||
codeman web --https # binds 127.0.0.1:3000
|
||||
tailscale serve --bg https / http://127.0.0.1:3000
|
||||
```
|
||||
|
||||
Only devices on your tailnet can reach it; Tailscale handles identity. No app
|
||||
password and no `0.0.0.0` bind required. (This is the maintainer's production
|
||||
setup.)
|
||||
|
||||
### B. Authenticated cloudflared tunnel + password
|
||||
|
||||
```bash
|
||||
export CODEMAN_PASSWORD=<password>
|
||||
codeman web --https
|
||||
cloudflared tunnel --url https://localhost:3000
|
||||
```
|
||||
|
||||
Always set `CODEMAN_PASSWORD` here — the tunnel connects over loopback, so the
|
||||
hook‑event exemption (§3) would otherwise be reachable from the public URL.
|
||||
|
||||
### C. Direct LAN bind + password
|
||||
|
||||
```bash
|
||||
export CODEMAN_PASSWORD=<password>
|
||||
codeman web --https --host 0.0.0.0
|
||||
```
|
||||
|
||||
Exposes the port on all interfaces; the password is the only thing protecting it.
|
||||
|
||||
### Avoid
|
||||
|
||||
`--host 0.0.0.0` **without** a password. Codeman will start (and warn), but
|
||||
anyone on the network can control your Claude sessions. Never re‑expose `0.0.0.0`
|
||||
without a password.
|
||||
|
||||
---
|
||||
|
||||
## 5. File‑serving hardening
|
||||
|
||||
Three routes serve workspace files; all require a valid `sessionId` and run the
|
||||
shared path validator `validateSessionFilePath()` (`src/web/route-helpers.ts`):
|
||||
it `realpath`s the target **before** the boundary check and rejects anything that
|
||||
escapes the session working directory (`..`, absolute paths, and symlinks that
|
||||
resolve outside). The realpath‑before‑check ordering closes the validation‑time
|
||||
TOCTOU window.
|
||||
|
||||
| Route | Cap | Notes |
|
||||
|-------|-----|-------|
|
||||
| `file-content` | 10 MB | text preview |
|
||||
| `file-raw` | 50 MB | inline MIME map; **`X-Content-Type-Options: nosniff` on all responses** |
|
||||
| `POST /api/download` | 50 MB | forced `attachment`; sensitive‑path blocklist |
|
||||
|
||||
### SVG / content‑type XSS
|
||||
|
||||
A workspace `.svg` served inline as `image/svg+xml` is a stored‑XSS vector (SVG
|
||||
can carry `<script>`, same‑origin = full session control). `file-raw` therefore
|
||||
serves `.svg` as `application/octet-stream` + `Content-Disposition: attachment` +
|
||||
`nosniff`. With global `nosniff` + CSP `default-src 'self'`, other text types
|
||||
(`.html`, `.xml`, …) that fall through to `octet-stream` are not rendered as HTML
|
||||
either. Trusted QR/welcome SVGs are injected from API JSON (`innerHTML`), not via
|
||||
`file-raw`, so they are unaffected.
|
||||
|
||||
### Download sensitive‑path blocklist
|
||||
|
||||
`/api/download` additionally refuses a blocklist of sensitive paths
|
||||
(`/etc/shadow`, `~/.ssh/`, `.env`, `*credentials*`, `.aws/credentials`, …). This
|
||||
is **defense‑in‑depth, not the primary boundary** — the realpath containment is
|
||||
the control.
|
||||
|
||||
### Known limitation — `workingDir` scope
|
||||
|
||||
The file‑route boundary is the session's `workingDir`, and `POST /api/sessions`
|
||||
currently accepts an arbitrary absolute `workingDir` (validated as "exists + is a
|
||||
directory"). A session created with `workingDir=/` can therefore read files
|
||||
across the filesystem within that boundary. This is **pre‑existing** across all
|
||||
file routes and not widened by the recent changes. Recommended follow‑up:
|
||||
constrain `workingDir` to an allowlist (e.g. under the cases dir / `$HOME`).
|
||||
|
||||
---
|
||||
|
||||
## 6. tmux launch hardening (COD‑31)
|
||||
|
||||
New sessions and respawns launch the tmux server/pane from a stable `/tmp`
|
||||
(`TMUX_LAUNCH_CWD`) and then `cd` into the real workspace **inside** the pane,
|
||||
against the live mount table:
|
||||
|
||||
```
|
||||
respawn-pane -k -c /tmp -t <session> bash -c "cd <workingDir> && <cmd>"
|
||||
```
|
||||
|
||||
This avoids a class of failures on FUSE/rclone‑mounted workspaces where a
|
||||
transient mount blip at launch poisons tmux's long‑lived cwd and crashes
|
||||
`new-session`. Safety properties:
|
||||
|
||||
- **Fail‑safe cwd:** the command is `cd "<dir>" && <cmd>` — if `cd` fails the CLI
|
||||
does **not** run in `/tmp`; the pane dies with a visible error instead.
|
||||
- **No injection:** `workingDir` passes `isValidWorkingDir` (absolute, rejects
|
||||
`;&|$\`(){}<>'"` and newlines and `..`) and `isValidPath`, and is double‑quoted
|
||||
in the pane command. Paths with spaces work; metacharacters are rejected before
|
||||
reaching the shell.
|
||||
- It does not change which tmux socket is targeted, so instance isolation (§8) is
|
||||
preserved.
|
||||
|
||||
---
|
||||
|
||||
## 7. Supply‑chain & build‑asset hardening (COD‑28)
|
||||
|
||||
- **Dependency advisories:** security‑sensitive ranges are bumped to patched
|
||||
versions, and `overrides` force patched transitive deps (`picomatch`,
|
||||
`basic-ftp`, `fast-uri`, `flatted`). `test/dependency-security.test.ts` asserts
|
||||
these stay patched in the lockfile.
|
||||
- **Lockfile integrity:** `npm run check:lockfile` (CI on every push/PR) fails on
|
||||
drift between `package.json` and `package-lock.json`. All lockfile entries
|
||||
resolve to `registry.npmjs.org` with `sha512` integrity hashes.
|
||||
- **Public‑asset checker:** `npm run check:public-assets`
|
||||
(`scripts/check-public-assets.mjs`) scans `src/web/public/**` for literal NUL
|
||||
bytes and runs `node --check` on every `.js` file (syntax validation), plus a
|
||||
Prettier pass on maintained files. It uses `execFileSync` with argv arrays (no
|
||||
shell), so filenames/content cannot inject commands; `node --check` only parses,
|
||||
never executes. Large hand‑formatted/generated assets (`app.js`, the gesture
|
||||
bundle, vendored libs) are `.prettierignore`d for the style pass, but the NUL +
|
||||
syntax checks still cover them.
|
||||
|
||||
---
|
||||
|
||||
## 8. Multi‑instance isolation
|
||||
|
||||
The tmux socket (`tmux -L codeman[-<instance>]`) and data dir
|
||||
(`~/.codeman[-<instance>]`) are **process‑wide and shared by every Codeman on the
|
||||
machine**, derived from `CODEMAN_INSTANCE` (`src/config/instance.ts`). A second
|
||||
instance on the **same** socket discovers and attaches PTYs to the first
|
||||
instance's live sessions. To run instances side by side, give each a distinct
|
||||
`CODEMAN_INSTANCE` (scopes both dir + socket), or set `CODEMAN_TMUX_SOCKET` +
|
||||
`CODEMAN_DATA_DIR` individually. `CODEMAN_INSTANCE` defaults to empty = the
|
||||
production layout (`~/.codeman`, `-L codeman`, port 3000).
|
||||
|
||||
---
|
||||
|
||||
## 9. Transport security headers
|
||||
|
||||
`registerSecurityHeaders` applies on every response:
|
||||
|
||||
- `Content-Security-Policy: default-src 'self'` (widened only for `/gesture/`
|
||||
assets when `CODEMAN_GESTURE=1`, to load self‑hosted MediaPipe)
|
||||
- `X-Content-Type-Options: nosniff`
|
||||
- `X-Frame-Options`
|
||||
- `Strict-Transport-Security` when served over HTTPS
|
||||
- CORS restricted to localhost origins
|
||||
|
||||
---
|
||||
|
||||
## 10. Quick reference
|
||||
|
||||
| Env / flag | Effect |
|
||||
|------------|--------|
|
||||
| `CODEMAN_PASSWORD` (+ `CODEMAN_USERNAME`) | Enable HTTP Basic auth |
|
||||
| `--host` / `CODEMAN_HOST` | Bind host (default `127.0.0.1`) |
|
||||
| `--allow-unauthenticated-network` / `CODEMAN_ALLOW_UNAUTHENTICATED_NETWORK` | Acknowledge an unauthenticated non‑loopback bind (downgrades the warning) |
|
||||
| `--https` | Enable TLS (adds HSTS) |
|
||||
| `CODEMAN_INSTANCE` | Scope tmux socket + data dir for isolation |
|
||||
| `CODEMAN_GESTURE=1` | Make the gesture overlay available (widens CSP) |
|
||||
|
||||
**Audit log:** session lifecycle and server start are recorded in
|
||||
`~/.codeman/session-lifecycle.jsonl`.
|
||||
@@ -7,7 +7,7 @@
|
||||
# Environment variables:
|
||||
# CODEMAN_NONINTERACTIVE=1 - Skip all prompts (for CI/automation)
|
||||
# CODEMAN_INSTALL_DIR - Custom install directory (default: ~/.codeman/app)
|
||||
# CODEMAN_SKIP_SYSTEMD=1 - Skip systemd service setup prompt
|
||||
# CODEMAN_SKIP_SYSTEMD=1 - Skip systemd/launchd service setup prompt
|
||||
# CODEMAN_NODE_VERSION - Node.js major version to install (default: 22)
|
||||
# CODEMAN_REPO_URL - Custom git repository URL (default: upstream Codeman)
|
||||
# CODEMAN_BRANCH - Git branch to install (default: master)
|
||||
@@ -353,8 +353,15 @@ ensure_sudo() {
|
||||
die "sudo is required but not installed. Please install packages manually or run as root."
|
||||
fi
|
||||
# Validate sudo access
|
||||
if ! sudo -v 2>/dev/null; then
|
||||
die "Failed to obtain sudo privileges."
|
||||
# When piped (curl | bash), stdin is the pipe — redirect from /dev/tty so sudo can prompt
|
||||
if [[ -e /dev/tty ]]; then
|
||||
if ! sudo -v 2>/dev/null < /dev/tty; then
|
||||
die "Failed to obtain sudo privileges."
|
||||
fi
|
||||
else
|
||||
if ! sudo -v 2>/dev/null; then
|
||||
die "Failed to obtain sudo privileges. Try running the script directly instead of piping."
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
@@ -372,7 +379,12 @@ ensure_homebrew() {
|
||||
fi
|
||||
|
||||
info "Installing Homebrew first..."
|
||||
/bin/bash -c "$(download_to_stdout https://raw.githubusercontent.com/Homebrew/install/HEAD/install.sh)"
|
||||
# When piped (curl | bash), stdin is the pipe — Homebrew needs TTY for sudo password prompt
|
||||
if [[ -e /dev/tty ]]; then
|
||||
/bin/bash -c "$(download_to_stdout https://raw.githubusercontent.com/Homebrew/install/HEAD/install.sh)" < /dev/tty
|
||||
else
|
||||
NONINTERACTIVE=1 /bin/bash -c "$(download_to_stdout https://raw.githubusercontent.com/Homebrew/install/HEAD/install.sh)"
|
||||
fi
|
||||
|
||||
# Add Homebrew to PATH for Apple Silicon
|
||||
if [[ -f /opt/homebrew/bin/brew ]]; then
|
||||
@@ -787,9 +799,84 @@ setup_sc_alias() {
|
||||
}
|
||||
|
||||
# ============================================================================
|
||||
# Systemd Service Setup (Linux only)
|
||||
# Service Setup (Linux systemd / macOS launchd)
|
||||
# ============================================================================
|
||||
|
||||
setup_launchd_service() {
|
||||
local plist_label="com.codeman.web"
|
||||
local agent_dir="$HOME/Library/LaunchAgents"
|
||||
local agent_plist="$agent_dir/$plist_label.plist"
|
||||
local daemon_plist="/Library/LaunchDaemons/$plist_label.plist"
|
||||
|
||||
info "Setting up macOS LaunchAgent..."
|
||||
|
||||
# Remove any existing LaunchDaemon (system-level) to prevent duplicates.
|
||||
# We standardize on LaunchAgent (user-level) — it doesn't require sudo,
|
||||
# inherits the user's environment, and is the correct choice for user apps.
|
||||
if [[ -f "$daemon_plist" ]]; then
|
||||
warn "Found system-level LaunchDaemon at $daemon_plist — removing to prevent duplicate"
|
||||
sudo launchctl unload "$daemon_plist" 2>/dev/null || true
|
||||
sudo rm -f "$daemon_plist"
|
||||
success "Removed duplicate LaunchDaemon"
|
||||
fi
|
||||
|
||||
# Unload existing agent before overwriting
|
||||
if [[ -f "$agent_plist" ]]; then
|
||||
launchctl unload "$agent_plist" 2>/dev/null || true
|
||||
fi
|
||||
|
||||
mkdir -p "$agent_dir"
|
||||
|
||||
# Build PATH: ensure /opt/homebrew/bin (Apple Silicon) and ~/.local/bin are included
|
||||
local svc_path="/opt/homebrew/bin:/usr/local/bin:$HOME/.local/bin:/usr/bin:/bin:/usr/sbin:/sbin"
|
||||
|
||||
# Find node binary path
|
||||
local node_path
|
||||
node_path=$(command -v node)
|
||||
|
||||
cat > "$agent_plist" << EOF
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
|
||||
<plist version="1.0">
|
||||
<dict>
|
||||
<key>Label</key>
|
||||
<string>$plist_label</string>
|
||||
<key>ProgramArguments</key>
|
||||
<array>
|
||||
<string>$node_path</string>
|
||||
<string>$INSTALL_DIR/dist/index.js</string>
|
||||
<string>web</string>
|
||||
</array>
|
||||
<key>EnvironmentVariables</key>
|
||||
<dict>
|
||||
<key>PATH</key>
|
||||
<string>$svc_path</string>
|
||||
<key>HOME</key>
|
||||
<string>$HOME</string>
|
||||
<key>LANG</key>
|
||||
<string>en_US.UTF-8</string>
|
||||
</dict>
|
||||
<key>WorkingDirectory</key>
|
||||
<string>$HOME</string>
|
||||
<key>RunAtLoad</key>
|
||||
<true/>
|
||||
<key>KeepAlive</key>
|
||||
<true/>
|
||||
<key>ThrottleInterval</key>
|
||||
<integer>10</integer>
|
||||
<key>StandardOutPath</key>
|
||||
<string>/tmp/codeman.log</string>
|
||||
<key>StandardErrorPath</key>
|
||||
<string>/tmp/codeman.log</string>
|
||||
</dict>
|
||||
</plist>
|
||||
EOF
|
||||
|
||||
launchctl load "$agent_plist" 2>/dev/null || true
|
||||
|
||||
success "LaunchAgent installed and started"
|
||||
}
|
||||
|
||||
setup_systemd_service() {
|
||||
local service_dir="$HOME/.config/systemd/user"
|
||||
local service_file="$service_dir/codeman-web.service"
|
||||
@@ -1139,17 +1226,25 @@ main() {
|
||||
echo ""
|
||||
|
||||
local launch_choice=""
|
||||
local has_systemd=false
|
||||
local has_service=false
|
||||
local service_type=""
|
||||
|
||||
if [[ "$os" == "linux" ]] && [[ "$SKIP_SYSTEMD" != "1" ]] && command -v systemctl &>/dev/null; then
|
||||
has_systemd=true
|
||||
has_service=true
|
||||
service_type="systemd"
|
||||
elif [[ "$os" == "macos" ]] && [[ "$SKIP_SYSTEMD" != "1" ]]; then
|
||||
has_service=true
|
||||
service_type="launchd"
|
||||
fi
|
||||
|
||||
if [[ "$has_systemd" == "true" ]]; then
|
||||
if [[ "$has_service" == "true" ]]; then
|
||||
local service_label="systemd service"
|
||||
[[ "$service_type" == "launchd" ]] && service_label="LaunchAgent"
|
||||
|
||||
echo -e " ${BOLD}How would you like to run Codeman?${NC}"
|
||||
echo ""
|
||||
echo -e " ${CYAN}1)${NC} Run now in this terminal"
|
||||
echo -e " ${CYAN}2)${NC} Install as systemd service (auto-start on boot)"
|
||||
echo -e " ${CYAN}2)${NC} Install as $service_label (auto-start on boot)"
|
||||
echo -e " ${CYAN}3)${NC} Don't start — I'll run it later"
|
||||
echo ""
|
||||
|
||||
@@ -1166,7 +1261,7 @@ main() {
|
||||
done
|
||||
fi
|
||||
else
|
||||
# macOS or no systemd — only offer run now or skip
|
||||
# No service manager available — only offer run now or skip
|
||||
echo -e " ${BOLD}Would you like to start Codeman now?${NC}"
|
||||
echo ""
|
||||
echo -e " ${CYAN}1)${NC} Run now in this terminal"
|
||||
@@ -1192,12 +1287,16 @@ main() {
|
||||
|
||||
echo ""
|
||||
|
||||
# Handle systemd setup
|
||||
# Handle service setup
|
||||
if [[ "$launch_choice" == "2" ]]; then
|
||||
setup_systemd_service
|
||||
if [[ "$service_type" == "launchd" ]]; then
|
||||
setup_launchd_service
|
||||
else
|
||||
setup_systemd_service
|
||||
fi
|
||||
|
||||
# Offer tunnel service if cloudflared is available
|
||||
if check_cloudflared && [[ -f "$INSTALL_DIR/scripts/codeman-tunnel.service" ]]; then
|
||||
# Offer tunnel service if cloudflared is available (Linux only — systemd tunnel service)
|
||||
if [[ "$service_type" == "systemd" ]] && check_cloudflared && [[ -f "$INSTALL_DIR/scripts/codeman-tunnel.service" ]]; then
|
||||
echo ""
|
||||
if prompt_yes_no "Also set up Cloudflare tunnel service? (requires CODEMAN_PASSWORD)" "n"; then
|
||||
setup_tunnel_service
|
||||
@@ -1212,10 +1311,16 @@ main() {
|
||||
echo ""
|
||||
echo -e " ${BOLD}Manage the service:${NC}"
|
||||
echo ""
|
||||
echo -e " ${CYAN}systemctl --user stop codeman-web${NC} # Stop"
|
||||
echo -e " ${CYAN}systemctl --user restart codeman-web${NC} # Restart"
|
||||
echo -e " ${CYAN}systemctl --user status codeman-web${NC} # Check status"
|
||||
echo -e " ${CYAN}journalctl --user -u codeman-web -f${NC} # View logs"
|
||||
if [[ "$service_type" == "launchd" ]]; then
|
||||
echo -e " ${CYAN}launchctl unload ~/Library/LaunchAgents/com.codeman.web.plist${NC} # Stop"
|
||||
echo -e " ${CYAN}launchctl load ~/Library/LaunchAgents/com.codeman.web.plist${NC} # Start"
|
||||
echo -e " ${CYAN}tail -f /tmp/codeman.log${NC} # View logs"
|
||||
else
|
||||
echo -e " ${CYAN}systemctl --user stop codeman-web${NC} # Stop"
|
||||
echo -e " ${CYAN}systemctl --user restart codeman-web${NC} # Restart"
|
||||
echo -e " ${CYAN}systemctl --user status codeman-web${NC} # Check status"
|
||||
echo -e " ${CYAN}journalctl --user -u codeman-web -f${NC} # View logs"
|
||||
fi
|
||||
echo ""
|
||||
fi
|
||||
|
||||
@@ -1289,11 +1394,17 @@ update() {
|
||||
success "Updated to $(node -e "console.log(require('./package.json').version)")"
|
||||
echo ""
|
||||
|
||||
# Auto-restart systemd service if it's running, otherwise tell the user
|
||||
if systemctl --user is-active codeman-web.service &>/dev/null; then
|
||||
# Auto-restart service if running, otherwise tell the user
|
||||
local agent_plist="$HOME/Library/LaunchAgents/com.codeman.web.plist"
|
||||
if systemctl --user is-active codeman-web.service &>/dev/null 2>&1; then
|
||||
info "Restarting codeman-web service..."
|
||||
systemctl --user restart codeman-web.service
|
||||
success "codeman-web service restarted"
|
||||
elif [[ -f "$agent_plist" ]]; then
|
||||
info "Restarting LaunchAgent..."
|
||||
launchctl unload "$agent_plist" 2>/dev/null || true
|
||||
launchctl load "$agent_plist" 2>/dev/null || true
|
||||
success "LaunchAgent restarted"
|
||||
else
|
||||
echo -e " ${DIM}Restart codeman web to use the new version:${NC}"
|
||||
echo -e " ${CYAN}pkill -f 'codeman.*web'; codeman web &${NC}"
|
||||
@@ -1306,9 +1417,9 @@ uninstall() {
|
||||
info "Uninstalling Codeman..."
|
||||
echo ""
|
||||
|
||||
# Stop and remove systemd services
|
||||
# Stop and remove systemd services (Linux)
|
||||
for svc in codeman-web codeman-tunnel; do
|
||||
if systemctl --user is-active "${svc}.service" &>/dev/null; then
|
||||
if systemctl --user is-active "${svc}.service" &>/dev/null 2>&1; then
|
||||
info "Stopping ${svc} service..."
|
||||
systemctl --user stop "${svc}.service"
|
||||
fi
|
||||
@@ -1324,6 +1435,20 @@ uninstall() {
|
||||
done
|
||||
systemctl --user daemon-reload 2>/dev/null || true
|
||||
|
||||
# Stop and remove launchd services (macOS)
|
||||
local agent_plist="$HOME/Library/LaunchAgents/com.codeman.web.plist"
|
||||
local daemon_plist="/Library/LaunchDaemons/com.codeman.web.plist"
|
||||
if [[ -f "$agent_plist" ]]; then
|
||||
launchctl unload "$agent_plist" 2>/dev/null || true
|
||||
rm -f "$agent_plist"
|
||||
success "Removed LaunchAgent"
|
||||
fi
|
||||
if [[ -f "$daemon_plist" ]]; then
|
||||
sudo launchctl unload "$daemon_plist" 2>/dev/null || true
|
||||
sudo rm -f "$daemon_plist"
|
||||
success "Removed LaunchDaemon"
|
||||
fi
|
||||
|
||||
# Remove symlinks
|
||||
local symlink_dir="$HOME/.local/bin"
|
||||
if [[ -L "$symlink_dir/codeman" ]]; then
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
{
|
||||
"$schema": "https://unpkg.com/knip@5/schema.json",
|
||||
"entry": [
|
||||
"scripts/*.mjs",
|
||||
"scripts/*.js",
|
||||
"scripts/watch-subagents.ts",
|
||||
"scripts/remotion/Root.tsx",
|
||||
"scripts/remotion/index.ts",
|
||||
"test/**/*.test.ts",
|
||||
"test/mobile/vitest.config.ts",
|
||||
"test/**/*.mjs"
|
||||
],
|
||||
"project": ["src/**/*.{ts,tsx}", "scripts/**/*.{ts,tsx,mjs,js}", "test/**/*.{ts,mjs}"],
|
||||
"ignoreExportsUsedInFile": true,
|
||||
"ignoreDependencies": ["@remotion/cli", "@remotion/transitions", "esbuild", "agent-browser"]
|
||||
}
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "aicodeman",
|
||||
"version": "0.4.3",
|
||||
"version": "0.9.1",
|
||||
"description": "The missing control plane for AI coding agents - run 20 autonomous agents with real-time monitoring and session persistence",
|
||||
"type": "module",
|
||||
"main": "dist/index.js",
|
||||
@@ -15,17 +15,20 @@
|
||||
"dev": "tsx src/index.ts web",
|
||||
"web": "node dist/index.js web",
|
||||
"clean": "rm -rf dist",
|
||||
"test": "vitest run",
|
||||
"test:watch": "vitest",
|
||||
"test:coverage": "vitest run --coverage",
|
||||
"test": "vitest run --config config/vitest.config.ts",
|
||||
"test:watch": "vitest --config config/vitest.config.ts",
|
||||
"test:coverage": "vitest run --config config/vitest.config.ts --coverage",
|
||||
"typecheck": "tsc --noEmit",
|
||||
"lint": "eslint 'src/**/*.ts'",
|
||||
"lint:fix": "eslint 'src/**/*.ts' --fix",
|
||||
"format": "prettier --write 'src/**/*.ts'",
|
||||
"format:check": "prettier --check 'src/**/*.ts'",
|
||||
"lint": "eslint --config config/eslint.config.js 'src/**/*.ts'",
|
||||
"lint:fix": "eslint --config config/eslint.config.js 'src/**/*.ts' --fix",
|
||||
"format": "prettier --write 'src/**/*.ts' 'src/web/public/**/*.{js,css,html,json}'",
|
||||
"format:check": "prettier --check 'src/**/*.ts' 'src/web/public/**/*.{js,css,html,json}'",
|
||||
"check:public-assets": "node scripts/check-public-assets.mjs",
|
||||
"capture:subagents": "node scripts/capture-subagent-screenshots.mjs",
|
||||
"changeset": "changeset",
|
||||
"version-packages": "changeset version",
|
||||
"version-packages": "changeset version && npm install --package-lock-only && node scripts/check-lockfile-sync.mjs",
|
||||
"check:lockfile": "node scripts/check-lockfile-sync.mjs",
|
||||
"knip": "npx --yes knip@latest",
|
||||
"release": "changeset publish"
|
||||
},
|
||||
"workspaces": [
|
||||
@@ -50,7 +53,8 @@
|
||||
"dependencies": {
|
||||
"@fastify/compress": "^8.3.1",
|
||||
"@fastify/cookie": "^11.0.2",
|
||||
"@fastify/static": "^8.0.0",
|
||||
"@fastify/multipart": "^10.0.0",
|
||||
"@fastify/static": "^9.1.3",
|
||||
"@fastify/websocket": "^11.2.0",
|
||||
"@xterm/addon-fit": "^0.11.0",
|
||||
"@xterm/addon-unicode11": "^0.9.0",
|
||||
@@ -59,18 +63,18 @@
|
||||
"chalk": "^5.3.0",
|
||||
"chokidar": "^3.6.0",
|
||||
"commander": "^12.1.0",
|
||||
"fastify": "^5.1.0",
|
||||
"fastify": "^5.8.5",
|
||||
"node-pty": "^1.1.0",
|
||||
"qrcode": "^1.5.4",
|
||||
"uuid": "^10.0.0",
|
||||
"uuid": "^14.0.0",
|
||||
"web-push": "^3.6.7",
|
||||
"zod": "^4.3.6"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@changesets/cli": "^2.29.8",
|
||||
"@eslint/js": "^9.0.0",
|
||||
"@remotion/cli": "4.0.429",
|
||||
"@remotion/transitions": "4.0.429",
|
||||
"@remotion/cli": "4.0.473",
|
||||
"@remotion/transitions": "4.0.473",
|
||||
"@types/node": "^20.19.33",
|
||||
"@types/pngjs": "^6.0.5",
|
||||
"@types/qrcode": "^1.5.6",
|
||||
@@ -78,7 +82,7 @@
|
||||
"@types/uuid": "^10.0.0",
|
||||
"@types/web-push": "^3.6.4",
|
||||
"@types/ws": "^8.18.1",
|
||||
"@vitest/coverage-v8": "^4.0.18",
|
||||
"@vitest/coverage-v8": "^4.1.8",
|
||||
"agent-browser": "^0.6.0",
|
||||
"esbuild": "^0.27.3",
|
||||
"eslint": "^9.0.0",
|
||||
@@ -87,16 +91,30 @@
|
||||
"pngjs": "^7.0.0",
|
||||
"prettier": "^3.4.0",
|
||||
"puppeteer": "^24.36.0",
|
||||
"remotion": "4.0.429",
|
||||
"remotion": "4.0.473",
|
||||
"tsx": "^4.15.0",
|
||||
"typescript": "^5.9.3",
|
||||
"typescript-eslint": "^8.0.0",
|
||||
"vitest": "^4.0.18"
|
||||
"vitest": "^4.1.8"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@remotion/compositor-linux-x64-gnu": "^4.0.432",
|
||||
"@rspack/binding-linux-x64-gnu": "^1.7.7"
|
||||
},
|
||||
"overrides": {
|
||||
"basic-ftp": "^5.3.1",
|
||||
"fast-uri": "^3.1.2",
|
||||
"flatted": "^3.4.2",
|
||||
"anymatch": {
|
||||
"picomatch": "^2.3.2"
|
||||
},
|
||||
"micromatch": {
|
||||
"picomatch": "^2.3.2"
|
||||
},
|
||||
"readdirp": {
|
||||
"picomatch": "^2.3.2"
|
||||
}
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=18.0.0"
|
||||
},
|
||||
|
||||
@@ -45,6 +45,6 @@
|
||||
"jsdom": "^24.1.3",
|
||||
"tsup": "^8.5.1",
|
||||
"typescript": "^5.5.0",
|
||||
"vitest": "^2.1.9"
|
||||
"vitest": "^4.1.8"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -19,6 +19,10 @@ const PORTS = {
|
||||
|
||||
const results = [];
|
||||
|
||||
function isCodemanTitle(title) {
|
||||
return typeof title === 'string' && title.startsWith('codeman:');
|
||||
}
|
||||
|
||||
function logSection(title) {
|
||||
console.log('\n' + '='.repeat(60));
|
||||
console.log(` ${title}`);
|
||||
@@ -88,7 +92,7 @@ async function main() {
|
||||
const page = await playwrightBrowser.newPage();
|
||||
await page.goto(`http://localhost:${PORTS.playwright}`);
|
||||
const title = await page.title();
|
||||
if (title !== 'Codeman') throw new Error(`Expected Codeman, got ${title}`);
|
||||
if (!isCodemanTitle(title)) throw new Error(`Expected codeman:<hostname>, got ${title}`);
|
||||
await page.close();
|
||||
});
|
||||
|
||||
@@ -149,7 +153,7 @@ async function main() {
|
||||
const page = await puppeteerBrowser.newPage();
|
||||
await page.goto(`http://localhost:${PORTS.puppeteer}`);
|
||||
const title = await page.title();
|
||||
if (title !== 'Codeman') throw new Error(`Expected Codeman, got ${title}`);
|
||||
if (!isCodemanTitle(title)) throw new Error(`Expected codeman:<hostname>, got ${title}`);
|
||||
await page.close();
|
||||
});
|
||||
|
||||
@@ -202,7 +206,7 @@ async function main() {
|
||||
agentBrowser(`open http://localhost:${PORTS.agentBrowser}`);
|
||||
await new Promise(r => setTimeout(r, 2000));
|
||||
const title = agentBrowserJson('get title');
|
||||
agentBrowserAvailable = title.title === 'Codeman';
|
||||
agentBrowserAvailable = isCodemanTitle(title.title);
|
||||
console.log(' Browser launched');
|
||||
|
||||
// Test 1: Page load
|
||||
@@ -210,7 +214,7 @@ async function main() {
|
||||
agentBrowser(`open http://localhost:${PORTS.agentBrowser}`);
|
||||
await new Promise(r => setTimeout(r, 1000));
|
||||
const title = agentBrowserJson('get title');
|
||||
if (title.title !== 'Codeman') throw new Error(`Expected Codeman, got ${title.title}`);
|
||||
if (!isCodemanTitle(title.title)) throw new Error(`Expected codeman:<hostname>, got ${title.title}`);
|
||||
});
|
||||
|
||||
// Test 2: Element selection
|
||||
|
||||
@@ -32,6 +32,9 @@ run('chmod dist/index.js', 'chmod +x dist/index.js');
|
||||
// 2. Copy static assets (clean first to remove stale hashed files from previous builds)
|
||||
run('clean public', 'rm -rf dist/web/public');
|
||||
run('prepare dirs', 'mkdir -p dist/web dist/templates dist/web/public/vendor');
|
||||
// Fetch the opt-in gesture overlay's MediaPipe wasm + model into src/ (idempotent,
|
||||
// non-fatal, kept out of git) so the copy below carries them into dist/.
|
||||
run('gesture assets', 'node scripts/fetch-gesture-assets.mjs');
|
||||
run('copy web assets', 'cp -r src/web/public dist/web/');
|
||||
run('copy template', 'cp src/templates/case-template.md dist/templates/');
|
||||
|
||||
@@ -93,6 +96,7 @@ console.log('\n[build] content-hash cache busting');
|
||||
'ralph-wizard.js',
|
||||
'api-client.js',
|
||||
'subagent-windows.js',
|
||||
'image-input.js',
|
||||
'vendor/xterm-zerolag-input.js',
|
||||
];
|
||||
const manifest = {};
|
||||
|
||||
@@ -428,7 +428,7 @@ const SUBAGENT_ACTIVITY = {
|
||||
'agent-002': [
|
||||
{ type: 'tool', tool: 'Glob', input: { pattern: 'test/**/*.test.ts' }, timestamp: new Date().toISOString(), agentId: 'agent-002' },
|
||||
{ type: 'tool', tool: 'Read', input: { file_path: '/home/arkon/codeman/test/respawn-test-utils.ts' }, timestamp: new Date().toISOString(), agentId: 'agent-002' },
|
||||
{ type: 'tool', tool: 'Read', input: { file_path: '/home/arkon/codeman/vitest.config.ts' }, timestamp: new Date().toISOString(), agentId: 'agent-002' },
|
||||
{ type: 'tool', tool: 'Read', input: { file_path: '/home/arkon/codeman/config/vitest.config.ts' }, timestamp: new Date().toISOString(), agentId: 'agent-002' },
|
||||
{ type: 'message', role: 'assistant', text: 'Analyzing test patterns: MockSession, unique ports, fileParallelism: false...', timestamp: new Date().toISOString(), agentId: 'agent-002' },
|
||||
],
|
||||
};
|
||||
|
||||
@@ -9,7 +9,7 @@
|
||||
*
|
||||
* Usage: node scripts/capture-video-screenshots.mjs
|
||||
* Port: 3198 (static file server)
|
||||
* Output: remotion/public/ (6 PNGs)
|
||||
* Output: scripts/scripts/remotion/public/ (6 PNGs)
|
||||
*/
|
||||
|
||||
import { chromium } from 'playwright';
|
||||
@@ -21,7 +21,7 @@ import { fileURLToPath } from 'url';
|
||||
const __dirname = fileURLToPath(new URL('.', import.meta.url));
|
||||
const PROJECT_ROOT = join(__dirname, '..');
|
||||
const PUBLIC_DIR = join(PROJECT_ROOT, 'src', 'web', 'public');
|
||||
const OUTPUT_DIR = join(PROJECT_ROOT, 'remotion', 'public');
|
||||
const OUTPUT_DIR = join(PROJECT_ROOT, 'scripts', 'remotion', 'public');
|
||||
const PORT = 3198;
|
||||
|
||||
const DESKTOP_VIEWPORT = { width: 1920, height: 1080 };
|
||||
@@ -516,7 +516,7 @@ async function captureDesktopWelcome(browser) {
|
||||
path: join(OUTPUT_DIR, 'desktop-welcome.png'),
|
||||
fullPage: false,
|
||||
});
|
||||
console.log(' Saved: remotion/public/desktop-welcome.png');
|
||||
console.log(' Saved: scripts/remotion/public/desktop-welcome.png');
|
||||
} finally {
|
||||
await context.close();
|
||||
}
|
||||
@@ -545,7 +545,7 @@ async function captureDesktopClaude(browser) {
|
||||
path: join(OUTPUT_DIR, 'desktop-claude.png'),
|
||||
fullPage: false,
|
||||
});
|
||||
console.log(' Saved: remotion/public/desktop-claude.png');
|
||||
console.log(' Saved: scripts/remotion/public/desktop-claude.png');
|
||||
} finally {
|
||||
await context.close();
|
||||
}
|
||||
@@ -575,7 +575,7 @@ async function captureDesktopBothClaude(browser) {
|
||||
path: join(OUTPUT_DIR, 'desktop-both-claude.png'),
|
||||
fullPage: false,
|
||||
});
|
||||
console.log(' Saved: remotion/public/desktop-both-claude.png');
|
||||
console.log(' Saved: scripts/remotion/public/desktop-both-claude.png');
|
||||
} finally {
|
||||
await context.close();
|
||||
}
|
||||
@@ -605,7 +605,7 @@ async function captureDesktopBothOpencode(browser) {
|
||||
path: join(OUTPUT_DIR, 'desktop-both-opencode.png'),
|
||||
fullPage: false,
|
||||
});
|
||||
console.log(' Saved: remotion/public/desktop-both-opencode.png');
|
||||
console.log(' Saved: scripts/remotion/public/desktop-both-opencode.png');
|
||||
} finally {
|
||||
await context.close();
|
||||
}
|
||||
@@ -634,7 +634,7 @@ async function captureMobileClaude(browser) {
|
||||
path: join(OUTPUT_DIR, 'mobile-claude.png'),
|
||||
fullPage: false,
|
||||
});
|
||||
console.log(' Saved: remotion/public/mobile-claude.png');
|
||||
console.log(' Saved: scripts/remotion/public/mobile-claude.png');
|
||||
} finally {
|
||||
await context.close();
|
||||
}
|
||||
@@ -663,7 +663,7 @@ async function captureMobileOpencode(browser) {
|
||||
path: join(OUTPUT_DIR, 'mobile-opencode.png'),
|
||||
fullPage: false,
|
||||
});
|
||||
console.log(' Saved: remotion/public/mobile-opencode.png');
|
||||
console.log(' Saved: scripts/remotion/public/mobile-opencode.png');
|
||||
} finally {
|
||||
await context.close();
|
||||
}
|
||||
@@ -706,12 +706,12 @@ async function main() {
|
||||
console.log('All 6 screenshots captured!');
|
||||
console.log('='.repeat(60));
|
||||
console.log('\nOutput files:');
|
||||
console.log(' remotion/public/desktop-welcome.png');
|
||||
console.log(' remotion/public/desktop-claude.png');
|
||||
console.log(' remotion/public/desktop-both-claude.png');
|
||||
console.log(' remotion/public/desktop-both-opencode.png');
|
||||
console.log(' remotion/public/mobile-claude.png');
|
||||
console.log(' remotion/public/mobile-opencode.png');
|
||||
console.log(' scripts/remotion/public/desktop-welcome.png');
|
||||
console.log(' scripts/remotion/public/desktop-claude.png');
|
||||
console.log(' scripts/remotion/public/desktop-both-claude.png');
|
||||
console.log(' scripts/remotion/public/desktop-both-opencode.png');
|
||||
console.log(' scripts/remotion/public/mobile-claude.png');
|
||||
console.log(' scripts/remotion/public/mobile-opencode.png');
|
||||
} catch (err) {
|
||||
console.error('\nFatal error:', err.message);
|
||||
console.error(err.stack);
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
#!/usr/bin/env node
|
||||
// Fails if package-lock.json's version fields don't match package.json.
|
||||
// Changesets bumps package.json but NOT the lockfile — this catches that drift
|
||||
// (the top-level `version` in lockfiles is metadata, so `npm ci` won't flag it).
|
||||
|
||||
import { readFileSync } from 'node:fs';
|
||||
import { resolve, dirname } from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
const repoRoot = resolve(dirname(fileURLToPath(import.meta.url)), '..');
|
||||
const pkg = JSON.parse(readFileSync(resolve(repoRoot, 'package.json'), 'utf8'));
|
||||
const lock = JSON.parse(readFileSync(resolve(repoRoot, 'package-lock.json'), 'utf8'));
|
||||
|
||||
const expected = pkg.version;
|
||||
const rootVersion = lock.version;
|
||||
const selfVersion = lock.packages?.['']?.version;
|
||||
|
||||
const mismatches = [];
|
||||
if (rootVersion !== expected) mismatches.push(` package-lock.json#.version = ${rootVersion} (expected ${expected})`);
|
||||
if (selfVersion !== expected) mismatches.push(` package-lock.json#.packages[""].version = ${selfVersion} (expected ${expected})`);
|
||||
|
||||
if (mismatches.length > 0) {
|
||||
console.error(`\nLockfile version drift detected (package.json is ${expected}):`);
|
||||
console.error(mismatches.join('\n'));
|
||||
console.error('\nFix: run `npm install --package-lock-only` and commit the updated package-lock.json.\n');
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
console.log(`Lockfile in sync with package.json (${expected}).`);
|
||||
@@ -0,0 +1,66 @@
|
||||
#!/usr/bin/env node
|
||||
|
||||
import { execFileSync } from 'node:child_process';
|
||||
import { readdirSync, readFileSync } from 'node:fs';
|
||||
import { dirname, extname, join, relative, resolve } from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
const repoRoot = resolve(dirname(fileURLToPath(import.meta.url)), '..');
|
||||
const publicRoot = resolve(repoRoot, 'src/web/public');
|
||||
const prettierBin = resolve(repoRoot, 'node_modules/.bin/prettier');
|
||||
const checkedExtensions = new Set(['.js', '.css', '.html', '.json']);
|
||||
|
||||
function collectTextAssets(dir) {
|
||||
const files = [];
|
||||
for (const entry of readdirSync(dir, { withFileTypes: true })) {
|
||||
const fullPath = join(dir, entry.name);
|
||||
if (entry.isDirectory()) {
|
||||
files.push(...collectTextAssets(fullPath));
|
||||
continue;
|
||||
}
|
||||
if (checkedExtensions.has(extname(entry.name))) {
|
||||
files.push(fullPath);
|
||||
}
|
||||
}
|
||||
return files;
|
||||
}
|
||||
|
||||
function findNullByte(buffer) {
|
||||
for (let i = 0; i < buffer.length; i += 1) {
|
||||
if (buffer[i] === 0) return i;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
const files = collectTextAssets(publicRoot);
|
||||
const failures = [];
|
||||
|
||||
for (const file of files) {
|
||||
const rel = relative(repoRoot, file);
|
||||
const data = readFileSync(file);
|
||||
const nullByteIndex = findNullByte(data);
|
||||
if (nullByteIndex !== -1) {
|
||||
failures.push(`${rel}: contains literal NUL byte at offset ${nullByteIndex}`);
|
||||
}
|
||||
|
||||
if (extname(file) === '.js') {
|
||||
try {
|
||||
execFileSync(process.execPath, ['--check', file], { cwd: repoRoot, stdio: 'pipe' });
|
||||
} catch (err) {
|
||||
failures.push(`${rel}: JavaScript syntax check failed\n${String(err.stderr || err.message).trim()}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
execFileSync(prettierBin, ['--check', ...files], { cwd: repoRoot, stdio: 'pipe' });
|
||||
} catch (err) {
|
||||
failures.push(`Prettier public asset check failed\n${String(err.stdout || err.stderr || err.message).trim()}`);
|
||||
}
|
||||
|
||||
if (failures.length > 0) {
|
||||
console.error(failures.join('\n\n'));
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
console.log(`Public asset checks passed (${files.length} files).`);
|
||||
@@ -0,0 +1,19 @@
|
||||
[Unit]
|
||||
Description=Codeman Cloudflare Named Tunnel
|
||||
After=network-online.target codeman-web.service
|
||||
Wants=network-online.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
ExecStart=/usr/bin/cloudflared tunnel --config %h/.cloudflared/codeman.yml run codeman
|
||||
Restart=always
|
||||
RestartSec=5
|
||||
KillMode=process
|
||||
|
||||
# Logging
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=codeman-tunnel-named
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -12,6 +12,14 @@ KillMode=process
|
||||
Environment=NODE_ENV=production
|
||||
Environment=HOME=/home/arkon
|
||||
Environment=NODE_COMPILE_CACHE=/home/arkon/.codeman/compile-cache
|
||||
# Loopback bind (default, no --host) + no password: safe out of the box. Hooks
|
||||
# reach 127.0.0.1, and `tailscale serve` fronts it on the tailnet only (real
|
||||
# cert, no LAN exposure, no app login). To expose on the LAN instead, add
|
||||
# Environment=CODEMAN_HOST=0.0.0.0 + Environment=CODEMAN_PASSWORD=... .
|
||||
# See docs/security-architecture.md.
|
||||
Environment=CODEMAN_GESTURE=1
|
||||
# ^ Makes the gesture-control overlay AVAILABLE (CSP widening + /gesture/ assets);
|
||||
# the actual on/off stays the per-user App Settings toggle (default OFF).
|
||||
|
||||
# Logging
|
||||
StandardOutput=journal
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
/**
|
||||
* @fileoverview Fetch the gesture-overlay runtime assets (MediaPipe wasm + the
|
||||
* gesture-recognizer model) into src/web/public/gesture/ so Codeman can serve
|
||||
* them same-origin (a browser content-blocker otherwise blocks the public CDNs
|
||||
* and the overlay fails to start). These are large binaries (~27 MB) kept OUT of
|
||||
* git (ignored explicitly via `src/web/public/gesture/wasm/` + `*.task` in
|
||||
* .gitignore); they are fetched here at install (postinstall) and build time.
|
||||
*
|
||||
* Idempotent: skips files already present. Non-fatal: the gesture overlay is
|
||||
* opt-in (CODEMAN_GESTURE=1), so a fetch failure only warns — it must not break
|
||||
* `npm install` / `npm run build`. The build then copies src/web/public into
|
||||
* dist/ as usual, so prod gets these too.
|
||||
*
|
||||
* The @mediapipe/tasks-vision version MUST match the one bundled into the gesture
|
||||
* overlay (Ark0N/codeman-gesture-control) so the wasm loader matches its JS API.
|
||||
*/
|
||||
import { mkdirSync, existsSync, statSync, writeFileSync } from 'node:fs';
|
||||
import { join, dirname } from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
const GESTURE = join(__dirname, '..', 'src', 'web', 'public', 'gesture');
|
||||
const WASM = join(GESTURE, 'wasm');
|
||||
|
||||
const MP_VERSION = '0.10.21'; // keep in sync with the gesture overlay's @mediapipe/tasks-vision
|
||||
const WASM_BASE = `https://cdn.jsdelivr.net/npm/@mediapipe/tasks-vision@${MP_VERSION}/wasm`;
|
||||
const MODEL_URL =
|
||||
'https://storage.googleapis.com/mediapipe-models/gesture_recognizer/gesture_recognizer/float16/1/gesture_recognizer.task';
|
||||
|
||||
const ASSETS = [
|
||||
{ url: `${WASM_BASE}/vision_wasm_internal.js`, path: join(WASM, 'vision_wasm_internal.js') },
|
||||
{ url: `${WASM_BASE}/vision_wasm_internal.wasm`, path: join(WASM, 'vision_wasm_internal.wasm') },
|
||||
{ url: `${WASM_BASE}/vision_wasm_nosimd_internal.js`, path: join(WASM, 'vision_wasm_nosimd_internal.js') },
|
||||
{ url: `${WASM_BASE}/vision_wasm_nosimd_internal.wasm`, path: join(WASM, 'vision_wasm_nosimd_internal.wasm') },
|
||||
{ url: MODEL_URL, path: join(GESTURE, 'gesture_recognizer.task') },
|
||||
];
|
||||
|
||||
async function main() {
|
||||
mkdirSync(WASM, { recursive: true });
|
||||
let fetched = 0;
|
||||
let skipped = 0;
|
||||
for (const a of ASSETS) {
|
||||
if (existsSync(a.path) && statSync(a.path).size > 0) {
|
||||
skipped++;
|
||||
continue;
|
||||
}
|
||||
const res = await fetch(a.url);
|
||||
if (!res.ok) throw new Error(`HTTP ${res.status} for ${a.url}`);
|
||||
writeFileSync(a.path, Buffer.from(await res.arrayBuffer()));
|
||||
fetched++;
|
||||
}
|
||||
console.log(`[gesture] MediaPipe assets ready (${fetched} fetched, ${skipped} cached) → ${GESTURE}`);
|
||||
}
|
||||
|
||||
main().catch((err) => {
|
||||
// Non-fatal: opt-in feature. Warn and exit 0 so install/build still succeed.
|
||||
console.warn(`[gesture] could not fetch MediaPipe assets — overlay disabled until fetched: ${err.message}`);
|
||||
});
|
||||
@@ -312,6 +312,20 @@ if (isGlobalInstall) {
|
||||
}
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// 4b. Fetch gesture-overlay runtime assets (MediaPipe wasm + model) for dev mode
|
||||
// (src/web/public/gesture/). Opt-in feature (CODEMAN_GESTURE=1); non-fatal.
|
||||
// Large binaries kept out of git; the build copies them into dist/.
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
if (!isGlobalInstall) {
|
||||
try {
|
||||
execSync(`node "${join(import.meta.dirname, 'fetch-gesture-assets.mjs')}"`, { stdio: 'inherit' });
|
||||
} catch {
|
||||
// Non-fatal — the gesture overlay is opt-in.
|
||||
}
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// 5. Install git pre-commit hook (format check)
|
||||
// ----------------------------------------------------------------------------
|
||||
@@ -446,3 +460,16 @@ if (process.env.CI || process.env.CODEMAN_NO_AUTOSTART) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Security note — printed on every install path
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
console.log(colors.bold('Security:'));
|
||||
console.log(colors.dim(' Codeman binds ') + colors.cyan('127.0.0.1') + colors.dim(' (this machine only) — no password needed by default.'));
|
||||
console.log(colors.dim(' To reach it from another device, do ONE of:'));
|
||||
console.log(colors.dim(' • ') + colors.cyan('tailscale serve') + colors.dim(' / ') + colors.cyan('cloudflared tunnel') + colors.dim(' (recommended), or'));
|
||||
console.log(colors.dim(' • ') + colors.cyan('codeman web --host 0.0.0.0') + colors.dim(' AND set ') + colors.cyan('CODEMAN_PASSWORD'));
|
||||
console.log(colors.dim(' A non-loopback bind without a password still starts, but warns loudly.'));
|
||||
console.log(colors.dim(' Details: docs/security-architecture.md'));
|
||||
console.log('');
|
||||
|
||||
|
Before Width: | Height: | Size: 41 KiB After Width: | Height: | Size: 41 KiB |
|
Before Width: | Height: | Size: 37 KiB After Width: | Height: | Size: 37 KiB |
|
Before Width: | Height: | Size: 39 KiB After Width: | Height: | Size: 39 KiB |
|
Before Width: | Height: | Size: 57 KiB After Width: | Height: | Size: 57 KiB |
|
Before Width: | Height: | Size: 390 KiB After Width: | Height: | Size: 390 KiB |
|
Before Width: | Height: | Size: 22 KiB After Width: | Height: | Size: 22 KiB |
@@ -0,0 +1,32 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# run-beta.sh — launch a BETA Codeman isolated from a production instance.
|
||||
#
|
||||
# Codeman's data dir (~/.codeman) and tmux socket (-L codeman) are process-wide
|
||||
# and shared by every instance on the machine. The code now DEFAULTS to that
|
||||
# production layout on port 3000 (safe for master / existing installs), so a beta
|
||||
# build no longer isolates itself automatically — this wrapper opts it in:
|
||||
#
|
||||
# CODEMAN_INSTANCE=beta → data dir ~/.codeman-beta + tmux socket codeman-beta
|
||||
# CODEMAN_PORT=5000 → listen on 5000 instead of 3000
|
||||
#
|
||||
# Result: the beta runs side-by-side with prod and can never discover/attach to
|
||||
# prod's live tmux sessions or clobber prod's state.json. Override either var to
|
||||
# run additional named instances, e.g. CODEMAN_INSTANCE=foo CODEMAN_PORT=5050.
|
||||
#
|
||||
# Usage: ./scripts/run-beta.sh [extra `codeman web` flags]
|
||||
# Build first (the beta runs the compiled dist): npm run build
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
export CODEMAN_INSTANCE="${CODEMAN_INSTANCE:-beta}"
|
||||
export CODEMAN_PORT="${CODEMAN_PORT:-5000}"
|
||||
|
||||
DIST="$(cd "$(dirname "$0")/.." && pwd)/dist/index.js"
|
||||
if [ ! -f "$DIST" ]; then
|
||||
echo "dist not found at $DIST — run 'npm run build' first." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Starting beta Codeman: instance='$CODEMAN_INSTANCE' (~/.codeman-$CODEMAN_INSTANCE, -L codeman-$CODEMAN_INSTANCE) on port $CODEMAN_PORT"
|
||||
exec node "$DIST" web "$@"
|
||||
@@ -0,0 +1,82 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# span-codeman.sh — open a Codeman window stretched across ALL displays, so that
|
||||
# in-page floating session panels can be dragged from one physical monitor to
|
||||
# the other. Spawned by the header "multi-monitor" button (POST
|
||||
# /api/system/span-displays), or run by hand at the desk.
|
||||
#
|
||||
# ── PREREQUISITE (one-time, manual) ──────────────────────────────────────────
|
||||
# System Settings → Desktop & Dock → turn OFF "Displays have separate Spaces",
|
||||
# then LOG OUT and back in. Until you do, macOS keeps every window on a single
|
||||
# display and this script's window will clamp to one monitor instead of spanning.
|
||||
# (Equivalent CLI: `defaults write com.apple.spaces spans-displays -bool true`,
|
||||
# still needs a re-login. Revert with `-bool false`.)
|
||||
#
|
||||
# Why a maximized --app window and not fullscreen: browser fullscreen is
|
||||
# per-display and will NOT span. We size a windowed app to the union of all
|
||||
# displays instead. macOS only.
|
||||
#
|
||||
# ── REMOTE CLIENT (Codeman server on another machine) ────────────────────────
|
||||
# The header "multi-monitor" button spawns this script SERVER-SIDE, so it opens
|
||||
# the window on the SERVER's displays and is gated to a macOS server. If your
|
||||
# Codeman runs elsewhere (e.g. a headless Linux box reached over Tailscale) and
|
||||
# YOUR monitors are on a Mac, the button can't help — the server can't open a
|
||||
# window on your machine. Instead, run this script LOCALLY on the Mac and pass
|
||||
# the remote URL as the argument:
|
||||
# ./span-codeman.sh "https://your-codeman.example.ts.net"
|
||||
# The osascript/browser launch all happen on the Mac; only the page is served
|
||||
# remotely, so the spanning window lands on your monitors. The "separate Spaces"
|
||||
# prerequisite above still applies on the Mac.
|
||||
#
|
||||
set -euo pipefail
|
||||
|
||||
URL="${1:-http://localhost:5000}"
|
||||
|
||||
# Union rect of all displays in top-left-origin points — exactly what Chromium's
|
||||
# --window-position/--window-size expect. Finder's desktop window bounds already
|
||||
# encloses every monitor (and handles a monitor placed left/above via a negative
|
||||
# origin), so no per-display math or coordinate flipping is needed.
|
||||
bounds=$(osascript -e 'tell application "Finder" to get bounds of window of desktop')
|
||||
X=$(echo "$bounds" | awk -F', *' '{print $1}')
|
||||
Y=$(echo "$bounds" | awk -F', *' '{print $2}')
|
||||
R=$(echo "$bounds" | awk -F', *' '{print $3}')
|
||||
B=$(echo "$bounds" | awk -F', *' '{print $4}')
|
||||
W=$((R - X))
|
||||
H=$((B - Y))
|
||||
echo "Display union: position ${X},${Y} size ${W}x${H}"
|
||||
|
||||
# Pick a Chromium-family browser. Brave leads the list — plain Google Chrome
|
||||
# bounced when launched this way on the desk machine (created its profile then
|
||||
# exited without a window). Force a specific one with, e.g.,
|
||||
# BROWSER="Google Chrome" ./span-codeman.sh
|
||||
app="${BROWSER:-}"
|
||||
if [ -z "$app" ]; then
|
||||
for c in "Brave Browser" "Google Chrome" "Google Chrome Beta" "Chromium" "Microsoft Edge"; do
|
||||
[ -x "/Applications/$c.app/Contents/MacOS/$c" ] && app="$c" && break
|
||||
done
|
||||
fi
|
||||
bin="/Applications/$app.app/Contents/MacOS/$app"
|
||||
[ -n "$app" ] && [ -x "$bin" ] || { echo "No Chrome-family browser found (BROWSER='$app')" >&2; exit 1; }
|
||||
|
||||
# A dedicated, PER-BROWSER profile forces a FRESH instance — an already-running
|
||||
# browser would hand the URL to itself and silently ignore the geometry flags.
|
||||
# Per-browser so a Chrome-made profile can't confuse Brave (or vice-versa).
|
||||
slug=$(echo "$app" | tr '[:upper:] ' '[:lower:]-')
|
||||
profile="$HOME/.codeman-gesture-$slug"
|
||||
|
||||
echo "Browser: $bin"
|
||||
echo "URL: $URL"
|
||||
|
||||
# Detach so the caller (terminal / web server) isn't blocked for the window's life.
|
||||
nohup "$bin" \
|
||||
--app="$URL" \
|
||||
--user-data-dir="$profile" \
|
||||
--window-position="${X},${Y}" \
|
||||
--window-size="${W},${H}" \
|
||||
--no-first-run \
|
||||
--no-default-browser-check \
|
||||
>/dev/null 2>&1 &
|
||||
|
||||
echo "Launched spanning window (pid $!)."
|
||||
echo "If it filled only one monitor, the 'separate Spaces' prerequisite above"
|
||||
echo "isn't active yet — toggle it off, log out/in, and re-run."
|
||||
@@ -32,6 +32,13 @@ set -e
|
||||
CODEMAN_STATE="$HOME/.codeman/state.json"
|
||||
CODEMAN_SESSIONS="$HOME/.codeman/mux-sessions.json"
|
||||
|
||||
# Dedicated tmux socket all Codeman sessions live on. MUST match
|
||||
# DEFAULT_CODEMAN_TMUX_SOCKET / CODEMAN_TMUX_SOCKET in src/tmux-manager.ts —
|
||||
# otherwise list-sessions would enumerate the user's default tmux server
|
||||
# (missing the real Codeman sessions, surfacing unrelated ones).
|
||||
CODEMAN_TMUX_SOCKET="${CODEMAN_TMUX_SOCKET:-codeman}"
|
||||
TMUX_CMD=(tmux -L "$CODEMAN_TMUX_SOCKET")
|
||||
|
||||
|
||||
# iPhone 17 Pro portrait width (conservative)
|
||||
MAX_WIDTH=44
|
||||
@@ -286,7 +293,7 @@ parse_sessions() {
|
||||
|
||||
# Get PID from tmux
|
||||
local pid
|
||||
pid=$(tmux display-message -t "$session_name" -p '#{pane_pid}' 2>/dev/null || echo "0")
|
||||
pid=$("${TMUX_CMD[@]}" display-message -t "$session_name" -p '#{pane_pid}' 2>/dev/null || echo "0")
|
||||
|
||||
SESSION_PIDS+=("$pid")
|
||||
MUX_NAMES+=("$session_name")
|
||||
@@ -314,7 +321,7 @@ parse_sessions() {
|
||||
fi
|
||||
|
||||
i=$((i + 1))
|
||||
done < <(tmux list-sessions 2>/dev/null || true)
|
||||
done < <("${TMUX_CMD[@]}" list-sessions 2>/dev/null || true)
|
||||
}
|
||||
|
||||
# ============================================================================
|
||||
@@ -462,7 +469,7 @@ attach_session() {
|
||||
echo -e "${D}(Ctrl+B D to detach)${R}"
|
||||
sleep 0.3
|
||||
|
||||
tmux attach-session -t "$mux_name"
|
||||
"${TMUX_CMD[@]}" attach-session -t "$mux_name"
|
||||
|
||||
return 0
|
||||
}
|
||||
|
||||
@@ -20,6 +20,13 @@ REVERSE='\033[7m'
|
||||
# Use the same path as codeman (src/tmux-manager.ts)
|
||||
SESSIONS_FILE="${HOME}/.codeman/mux-sessions.json"
|
||||
|
||||
# Dedicated tmux socket all Codeman sessions live on. MUST match
|
||||
# DEFAULT_CODEMAN_TMUX_SOCKET / CODEMAN_TMUX_SOCKET in src/tmux-manager.ts —
|
||||
# otherwise this script would talk to the user's default tmux server and never
|
||||
# see (or could mis-target) Codeman's sessions.
|
||||
CODEMAN_TMUX_SOCKET="${CODEMAN_TMUX_SOCKET:-codeman}"
|
||||
TMUX_CMD=(tmux -L "$CODEMAN_TMUX_SOCKET")
|
||||
|
||||
|
||||
# Cached data
|
||||
CACHED_JSON=""
|
||||
@@ -92,7 +99,7 @@ declare -A ALIVE_CACHE
|
||||
check_alive() {
|
||||
local mux_name=$1
|
||||
if [[ -z "${ALIVE_CACHE[$mux_name]+x}" ]]; then
|
||||
if tmux has-session -t "$mux_name" 2>/dev/null; then
|
||||
if "${TMUX_CMD[@]}" has-session -t "$mux_name" 2>/dev/null; then
|
||||
ALIVE_CACHE[$mux_name]=1
|
||||
else
|
||||
ALIVE_CACHE[$mux_name]=0
|
||||
@@ -111,8 +118,10 @@ kill_session() {
|
||||
local mux_name=$(get_session_field $idx "muxName")
|
||||
local pid=$(get_session_field $idx "pid")
|
||||
|
||||
# SAFETY: Never kill own tmux session
|
||||
local current_session=$(tmux display-message -p '#{session_name}' 2>/dev/null || echo "")
|
||||
# SAFETY: Never kill own tmux session. Queried on the Codeman socket; if run
|
||||
# from a session on a different socket this returns empty (no match), which
|
||||
# is fine — you can't be "inside" a Codeman-socket session you didn't attach to.
|
||||
local current_session=$("${TMUX_CMD[@]}" display-message -p '#{session_name}' 2>/dev/null || echo "")
|
||||
if [[ -n "$current_session" && "$mux_name" == "$current_session" ]]; then
|
||||
echo -e "${RED}BLOCKED: Cannot kill own tmux session: $mux_name${NC}"
|
||||
return 1
|
||||
@@ -120,7 +129,7 @@ kill_session() {
|
||||
|
||||
pkill -TERM -P $pid 2>/dev/null
|
||||
kill -TERM -$pid 2>/dev/null
|
||||
tmux kill-session -t "$mux_name" 2>/dev/null
|
||||
"${TMUX_CMD[@]}" kill-session -t "$mux_name" 2>/dev/null
|
||||
kill -KILL $pid 2>/dev/null
|
||||
|
||||
# Remove from JSON
|
||||
@@ -345,7 +354,7 @@ interactive_menu() {
|
||||
clear
|
||||
echo -e "${CYAN}Attaching... (Ctrl+B D to detach)${NC}"
|
||||
sleep 0.3
|
||||
tmux attach-session -t "$mux_name"
|
||||
"${TMUX_CMD[@]}" attach-session -t "$mux_name"
|
||||
tput civis
|
||||
need_full_redraw=1
|
||||
force_refresh
|
||||
@@ -474,7 +483,7 @@ main() {
|
||||
[[ -z "${2:-}" ]] && { echo "Usage: $0 attach <N>"; exit 1; }
|
||||
force_refresh
|
||||
local mux_name=$(get_session_field $(($2-1)) "muxName")
|
||||
check_alive "$mux_name" && tmux attach-session -t "$mux_name" || echo "Session dead or not found"
|
||||
check_alive "$mux_name" && "${TMUX_CMD[@]}" attach-session -t "$mux_name" || echo "Session dead or not found"
|
||||
;;
|
||||
kill)
|
||||
[[ -z "${2:-}" ]] && { echo "Usage: $0 kill <N|N,M|N-M>"; exit 1; }
|
||||
@@ -492,8 +501,8 @@ main() {
|
||||
;;
|
||||
kill-all)
|
||||
force_refresh
|
||||
# SAFETY: Never kill own tmux session
|
||||
local current_session=$(tmux display-message -p '#{session_name}' 2>/dev/null || echo "")
|
||||
# SAFETY: Never kill own tmux session (queried on the Codeman socket)
|
||||
local current_session=$("${TMUX_CMD[@]}" display-message -p '#{session_name}' 2>/dev/null || echo "")
|
||||
local killed=0
|
||||
for ((i=CACHED_COUNT-1; i>=0; i--)); do
|
||||
local mux_name=$(get_session_field $i "muxName")
|
||||
|
||||
@@ -1,45 +1,209 @@
|
||||
#!/usr/bin/env bash
|
||||
# Quick Cloudflare Tunnel for Codeman
|
||||
# Usage: ./scripts/tunnel.sh [start|stop|status|url]
|
||||
# Cloudflare Tunnel manager for Codeman
|
||||
# Usage: ./scripts/tunnel.sh [quick|named] [start|stop|status|url]
|
||||
#
|
||||
# Modes:
|
||||
# quick — Quick tunnel with random trycloudflare.com URL (default)
|
||||
# named — Named tunnel on a fixed hostname (requires setup, see below)
|
||||
#
|
||||
# Environment variables:
|
||||
# CLOUDFLARED_TUNNEL_NAME — tunnel name (default: codeman)
|
||||
# CLOUDFLARED_TUNNEL_ID — tunnel UUID (from: cloudflared tunnel list)
|
||||
# CODEMAN_TUNNEL_HOSTNAME — public hostname (e.g. codeman.example.com)
|
||||
#
|
||||
# First-time named tunnel setup:
|
||||
# cloudflared tunnel login
|
||||
# cloudflared tunnel create <tunnel-name>
|
||||
# cloudflared tunnel route dns <tunnel-name> <hostname>
|
||||
# ./scripts/tunnel.sh named setup # writes ~/.cloudflared/<tunnel-name>.yml
|
||||
set -euo pipefail
|
||||
|
||||
SERVICE="codeman-tunnel"
|
||||
QUICK_SERVICE="codeman-tunnel"
|
||||
NAMED_SERVICE="codeman-tunnel-named"
|
||||
TUNNEL_NAME="${CLOUDFLARED_TUNNEL_NAME:-codeman}"
|
||||
TUNNEL_HOSTNAME="${CODEMAN_TUNNEL_HOSTNAME:-codeman.example.com}"
|
||||
CODEMAN_PORT="3000"
|
||||
LOG_FILE="$HOME/.codeman/tunnel.log"
|
||||
|
||||
case "${1:-start}" in
|
||||
start)
|
||||
if ! systemctl --user is-active "$SERVICE" &>/dev/null; then
|
||||
# Install service if not already
|
||||
if ! systemctl --user cat "$SERVICE" &>/dev/null 2>&1; then
|
||||
cp "$(dirname "$0")/codeman-tunnel.service" "$HOME/.config/systemd/user/"
|
||||
systemctl --user daemon-reload
|
||||
fi
|
||||
systemctl --user start "$SERVICE"
|
||||
echo "Tunnel starting... waiting for URL"
|
||||
sleep 6
|
||||
fi
|
||||
# Extract the tunnel URL from journal
|
||||
URL=$(grep -oP 'https://[a-z0-9-]+\.trycloudflare\.com' "$HOME/.codeman/tunnel.log" 2>/dev/null | tail -1)
|
||||
if [ -n "$URL" ]; then
|
||||
echo "$URL"
|
||||
else
|
||||
echo "URL not ready yet, try: $0 url"
|
||||
fi
|
||||
# ── helpers ──────────────────────────────────────────────────────────────────
|
||||
|
||||
_require_cloudflared() {
|
||||
if ! command -v cloudflared &>/dev/null; then
|
||||
echo "Error: cloudflared not found. Install with: yay -S cloudflared" >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
_cloudflared_bin() {
|
||||
command -v cloudflared
|
||||
}
|
||||
|
||||
_install_service() {
|
||||
local svc_file="$1"
|
||||
local svc_name="$2"
|
||||
if ! systemctl --user cat "$svc_name" &>/dev/null 2>&1; then
|
||||
cp "$(dirname "$0")/$svc_file" "$HOME/.config/systemd/user/"
|
||||
systemctl --user daemon-reload
|
||||
echo "Service $svc_name installed."
|
||||
fi
|
||||
}
|
||||
|
||||
_install_named_service() {
|
||||
if ! systemctl --user cat "$NAMED_SERVICE" &>/dev/null 2>&1; then
|
||||
# Generate service file with the configured tunnel name
|
||||
sed "s/codeman\.yml/$TUNNEL_NAME.yml/g; s/run codeman/run $TUNNEL_NAME/g" \
|
||||
"$(dirname "$0")/codeman-tunnel-named.service" \
|
||||
> "$HOME/.config/systemd/user/codeman-tunnel-named.service"
|
||||
systemctl --user daemon-reload
|
||||
echo "Service $NAMED_SERVICE installed (tunnel: $TUNNEL_NAME)."
|
||||
fi
|
||||
}
|
||||
|
||||
# ── named tunnel setup ───────────────────────────────────────────────────────
|
||||
|
||||
_named_setup() {
|
||||
_require_cloudflared
|
||||
|
||||
local creds_dir="$HOME/.cloudflared"
|
||||
local config_file="$creds_dir/$TUNNEL_NAME.yml"
|
||||
# Replace with your tunnel ID (from: cloudflared tunnel list)
|
||||
local tunnel_id="${CLOUDFLARED_TUNNEL_ID:-YOUR_TUNNEL_ID_HERE}"
|
||||
local creds_file="$creds_dir/$tunnel_id.json"
|
||||
|
||||
if [ ! -f "$creds_file" ]; then
|
||||
echo "Credentials not found: $creds_file"
|
||||
echo "Run: cloudflared tunnel create $TUNNEL_NAME"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
cat > "$config_file" <<EOF
|
||||
tunnel: $tunnel_id
|
||||
credentials-file: $creds_file
|
||||
|
||||
ingress:
|
||||
- hostname: $TUNNEL_HOSTNAME
|
||||
service: http://localhost:$CODEMAN_PORT
|
||||
- service: http_status:404
|
||||
EOF
|
||||
|
||||
echo "Config written to $config_file"
|
||||
echo "Tunnel ID: $tunnel_id"
|
||||
echo "Hostname: $TUNNEL_HOSTNAME"
|
||||
echo ""
|
||||
echo "Next steps:"
|
||||
echo " 1. Add Cloudflare Access policy for $TUNNEL_HOSTNAME (Zero Trust dashboard)"
|
||||
echo " 2. ./scripts/tunnel.sh named start"
|
||||
}
|
||||
|
||||
# ── quick mode ───────────────────────────────────────────────────────────────
|
||||
|
||||
_quick_start() {
|
||||
if ! systemctl --user is-active "$QUICK_SERVICE" &>/dev/null; then
|
||||
_install_service "codeman-tunnel.service" "$QUICK_SERVICE"
|
||||
systemctl --user start "$QUICK_SERVICE"
|
||||
echo "Quick tunnel starting... waiting for URL"
|
||||
sleep 6
|
||||
fi
|
||||
local url
|
||||
url=$(grep -oP 'https://[a-z0-9-]+\.trycloudflare\.com' "$LOG_FILE" 2>/dev/null | tail -1)
|
||||
if [ -n "$url" ]; then
|
||||
echo "$url"
|
||||
else
|
||||
echo "URL not ready yet, try: $0 quick url"
|
||||
fi
|
||||
}
|
||||
|
||||
_quick_stop() {
|
||||
systemctl --user stop "$QUICK_SERVICE"
|
||||
echo "Quick tunnel stopped"
|
||||
}
|
||||
|
||||
_quick_status() {
|
||||
systemctl --user status "$QUICK_SERVICE" --no-pager 2>&1 | head -10
|
||||
echo ""
|
||||
echo "URL:"
|
||||
grep -oP 'https://[a-z0-9-]+\.trycloudflare\.com' "$LOG_FILE" 2>/dev/null | tail -1
|
||||
}
|
||||
|
||||
_quick_url() {
|
||||
grep -oP 'https://[a-z0-9-]+\.trycloudflare\.com' "$LOG_FILE" 2>/dev/null | tail -1
|
||||
}
|
||||
|
||||
# ── named mode ───────────────────────────────────────────────────────────────
|
||||
|
||||
_named_start() {
|
||||
_require_cloudflared
|
||||
if [ ! -f "$HOME/.cloudflared/$TUNNEL_NAME.yml" ]; then
|
||||
echo "Config not found. Run: $0 named setup"
|
||||
exit 1
|
||||
fi
|
||||
if ! systemctl --user is-active "$NAMED_SERVICE" &>/dev/null; then
|
||||
_install_named_service
|
||||
systemctl --user start "$NAMED_SERVICE"
|
||||
echo "Named tunnel starting..."
|
||||
sleep 3
|
||||
fi
|
||||
echo "https://$TUNNEL_HOSTNAME"
|
||||
}
|
||||
|
||||
_named_stop() {
|
||||
systemctl --user stop "$NAMED_SERVICE"
|
||||
echo "Named tunnel stopped"
|
||||
}
|
||||
|
||||
_named_status() {
|
||||
systemctl --user status "$NAMED_SERVICE" --no-pager 2>&1 | head -10
|
||||
echo ""
|
||||
echo "URL: https://$TUNNEL_HOSTNAME"
|
||||
}
|
||||
|
||||
_named_enable() {
|
||||
_install_named_service
|
||||
systemctl --user enable "$NAMED_SERVICE"
|
||||
echo "Named tunnel enabled at boot."
|
||||
}
|
||||
|
||||
_named_disable() {
|
||||
systemctl --user disable "$NAMED_SERVICE"
|
||||
echo "Named tunnel disabled."
|
||||
}
|
||||
|
||||
# ── dispatch ─────────────────────────────────────────────────────────────────
|
||||
|
||||
MODE="${1:-quick}"
|
||||
CMD="${2:-start}"
|
||||
|
||||
case "$MODE" in
|
||||
quick)
|
||||
case "$CMD" in
|
||||
start) _quick_start ;;
|
||||
stop) _quick_stop ;;
|
||||
status) _quick_status ;;
|
||||
url) _quick_url ;;
|
||||
*) echo "Usage: $0 quick [start|stop|status|url]"; exit 1 ;;
|
||||
esac
|
||||
;;
|
||||
stop)
|
||||
systemctl --user stop "$SERVICE"
|
||||
echo "Tunnel stopped"
|
||||
;;
|
||||
status)
|
||||
systemctl --user status "$SERVICE" --no-pager 2>&1 | head -10
|
||||
echo ""
|
||||
echo "URL:"
|
||||
grep -oP 'https://[a-z0-9-]+\.trycloudflare\.com' "$HOME/.codeman/tunnel.log" 2>/dev/null | tail -1
|
||||
;;
|
||||
url)
|
||||
grep -oP 'https://[a-z0-9-]+\.trycloudflare\.com' "$HOME/.codeman/tunnel.log" 2>/dev/null | tail -1
|
||||
named)
|
||||
case "$CMD" in
|
||||
start) _named_start ;;
|
||||
stop) _named_stop ;;
|
||||
status) _named_status ;;
|
||||
url) echo "https://$TUNNEL_HOSTNAME" ;;
|
||||
setup) _named_setup ;;
|
||||
enable) _named_enable ;;
|
||||
disable) _named_disable ;;
|
||||
*) echo "Usage: $0 named [start|stop|status|url|setup|enable|disable]"; exit 1 ;;
|
||||
esac
|
||||
;;
|
||||
# backward compat: no mode prefix → quick tunnel
|
||||
start) _quick_start ;;
|
||||
stop) _quick_stop ;;
|
||||
status) _quick_status ;;
|
||||
url) _quick_url ;;
|
||||
*)
|
||||
echo "Usage: $0 [start|stop|status|url]"
|
||||
echo "Usage: $0 [quick|named] [start|stop|status|url]"
|
||||
echo " $0 named setup # first-time named tunnel configuration"
|
||||
echo " $0 named enable # start at boot"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
@@ -30,6 +30,7 @@ import { join } from 'node:path';
|
||||
import { EventEmitter } from 'node:events';
|
||||
import { getAugmentedPath, ANSI_ESCAPE_PATTERN_SIMPLE } from './utils/index.js';
|
||||
import { AI_CHECK_MAX_BACKOFF_MS } from './config/ai-defaults.js';
|
||||
import { getErrorMessage } from './types.js';
|
||||
|
||||
// ========== Security Validation ==========
|
||||
|
||||
@@ -293,7 +294,7 @@ export abstract class AiCheckerBase<
|
||||
this.emit('checkCompleted', result);
|
||||
return result;
|
||||
} catch (err) {
|
||||
const errorMsg = err instanceof Error ? err.message : String(err);
|
||||
const errorMsg = getErrorMessage(err);
|
||||
this.handleError(errorMsg);
|
||||
const result = this.createErrorResult(errorMsg, Date.now() - this.checkStartTime);
|
||||
this.emit('checkFailed', errorMsg);
|
||||
@@ -412,9 +413,7 @@ export abstract class AiCheckerBase<
|
||||
});
|
||||
muxProcess.unref();
|
||||
} catch (err) {
|
||||
throw new Error(
|
||||
`Failed to spawn ${this.checkDescription} tmux session: ${err instanceof Error ? err.message : String(err)}`
|
||||
);
|
||||
throw new Error(`Failed to spawn ${this.checkDescription} tmux session: ${getErrorMessage(err)}`);
|
||||
}
|
||||
|
||||
// Poll the temp file for completion
|
||||
|
||||
@@ -41,7 +41,7 @@ import {
|
||||
|
||||
// ========== Types ==========
|
||||
|
||||
export type AiIdleCheckConfig = AiCheckerConfigBase;
|
||||
type AiIdleCheckConfig = AiCheckerConfigBase;
|
||||
|
||||
export type AiCheckVerdict = 'IDLE' | 'WORKING' | 'ERROR';
|
||||
|
||||
|
||||
@@ -40,13 +40,13 @@ import {
|
||||
|
||||
// ========== Types ==========
|
||||
|
||||
export type AiPlanCheckConfig = AiCheckerConfigBase;
|
||||
type AiPlanCheckConfig = AiCheckerConfigBase;
|
||||
|
||||
export type AiPlanCheckVerdict = 'PLAN_MODE' | 'NOT_PLAN_MODE' | 'ERROR';
|
||||
|
||||
export type AiPlanCheckResult = AiCheckerResultBase<AiPlanCheckVerdict>;
|
||||
|
||||
export type AiPlanCheckState = AiCheckerStateBase<AiPlanCheckVerdict>;
|
||||
type AiPlanCheckState = AiCheckerStateBase<AiPlanCheckVerdict>;
|
||||
|
||||
// ========== Constants ==========
|
||||
|
||||
@@ -64,20 +64,27 @@ const DEFAULT_PLAN_CHECK_CONFIG: AiPlanCheckConfig = {
|
||||
const VERDICT_PATTERN = /^\s*(PLAN_MODE|NOT_PLAN_MODE)\b/i;
|
||||
|
||||
/** The prompt sent to the AI plan checker */
|
||||
const AI_PLAN_CHECK_PROMPT = `Analyze this terminal output from a running Claude Code session. Determine if the terminal is currently showing a PLAN MODE APPROVAL PROMPT or not.
|
||||
const AI_PLAN_CHECK_PROMPT = `Analyze this terminal output from a running Claude Code session. Determine if the terminal is currently showing a NUMBERED SELECTION MENU that is waiting for the user to press Enter on the highlighted default option.
|
||||
|
||||
A plan mode approval prompt is a numbered selection menu that Claude Code shows when it wants the user to approve a plan before proceeding. It typically has these characteristics:
|
||||
A qualifying menu has all of these characteristics:
|
||||
- A numbered list of options (e.g., "1. Yes", "2. No", "3. Type your own")
|
||||
- A selection indicator arrow (❯ or >) pointing to one of the options
|
||||
- Text asking for approval like "Would you like to proceed?" or "Ready to implement?"
|
||||
- The prompt appears at the BOTTOM of the output (most recent content)
|
||||
- A selection indicator arrow (❯ or >) pointing to one of the options (the default)
|
||||
- The menu appears at the BOTTOM of the output (most recent content)
|
||||
- It is asking the user to choose, not just displaying numbered information
|
||||
|
||||
NOT a plan mode prompt:
|
||||
This includes BOTH:
|
||||
- Plan-mode approval prompts ("Would you like to proceed?" / "Ready to implement?")
|
||||
- AskUserQuestion / elicitation dialogs (Claude Code's numbered question menus)
|
||||
|
||||
NOT a qualifying menu:
|
||||
- Claude actively working (spinners, "Thinking", tool execution)
|
||||
- A completed response with no selection menu
|
||||
- An AskUserQuestion/elicitation dialog (different format, free-text input)
|
||||
- A completed response with no selection menu visible
|
||||
- A free-text input field with no numbered options
|
||||
- A numbered LIST in the assistant's prose with no selection arrow
|
||||
- Network lag or mid-output pause
|
||||
- Any state without a visible numbered selection menu
|
||||
- Any state without a visible selector arrow on a numbered option
|
||||
|
||||
The verdict name PLAN_MODE is historical — it now means "auto-accept this selection menu by pressing Enter on the default".
|
||||
|
||||
Terminal output (most recent at bottom):
|
||||
---
|
||||
|
||||
@@ -99,7 +99,7 @@ const LOG_FILE_MENTION_PATTERN = /([/~][^\s'"<>|;&\n]*(?:\.log|\.txt|\.out|\/log
|
||||
/**
|
||||
* Events emitted by BashToolParser.
|
||||
*/
|
||||
export interface BashToolParserEvents {
|
||||
interface BashToolParserEvents {
|
||||
/** New Bash tool with file paths started */
|
||||
toolStart: [tool: ActiveBashTool];
|
||||
/** Bash tool completed */
|
||||
@@ -111,7 +111,7 @@ export interface BashToolParserEvents {
|
||||
/**
|
||||
* Configuration options for BashToolParser.
|
||||
*/
|
||||
export interface BashToolParserConfig {
|
||||
interface BashToolParserConfig {
|
||||
/** Session ID this parser belongs to */
|
||||
sessionId: string;
|
||||
/** Whether the parser is enabled (default: true) */
|
||||
@@ -470,113 +470,91 @@ export class BashToolParser extends EventEmitter<BashToolParserEvents> {
|
||||
* Process a single pre-stripped line of terminal output.
|
||||
*/
|
||||
private processCleanLine(cleanLine: string): void {
|
||||
// Check for tool start
|
||||
if (this._handleToolStart(cleanLine)) return;
|
||||
if (this._handleToolCompletion(cleanLine)) return;
|
||||
if (this._handleTextCommand(cleanLine)) return;
|
||||
this._handleLogFileMention(cleanLine);
|
||||
}
|
||||
|
||||
private _handleToolStart(cleanLine: string): boolean {
|
||||
const startMatch = cleanLine.match(BASH_TOOL_START_PATTERN);
|
||||
if (startMatch) {
|
||||
const command = startMatch[1];
|
||||
const timeout = startMatch[2]?.trim();
|
||||
if (!startMatch) return false;
|
||||
|
||||
// Check if this is a file-viewing command
|
||||
if (this.isFileViewerCommand(command)) {
|
||||
const filePaths = this.extractFilePaths(command);
|
||||
const command = startMatch[1];
|
||||
const timeout = startMatch[2]?.trim();
|
||||
|
||||
// Skip if any file path is already tracked (cross-pattern dedup)
|
||||
if (filePaths.some((fp) => this.isFilePathTracked(fp))) {
|
||||
return;
|
||||
}
|
||||
if (!this.isFileViewerCommand(command)) return true;
|
||||
|
||||
if (filePaths.length > 0) {
|
||||
const tool: ActiveBashTool = {
|
||||
id: uuidv4(),
|
||||
command,
|
||||
filePaths,
|
||||
timeout,
|
||||
startedAt: Date.now(),
|
||||
status: 'running',
|
||||
sessionId: this._sessionId,
|
||||
};
|
||||
const filePaths = this.extractFilePaths(command);
|
||||
|
||||
// Enforce max tools limit
|
||||
if (this._activeTools.size >= MAX_ACTIVE_TOOLS) {
|
||||
// Remove oldest tool
|
||||
const oldest = Array.from(this._activeTools.entries()).sort((a, b) => a[1].startedAt - b[1].startedAt)[0];
|
||||
if (oldest) {
|
||||
this._activeTools.delete(oldest[0]);
|
||||
}
|
||||
// Skip if any file path is already tracked (cross-pattern dedup)
|
||||
if (filePaths.some((fp) => this.isFilePathTracked(fp))) return true;
|
||||
|
||||
if (filePaths.length > 0) {
|
||||
const tool = this._createActiveTool(command, filePaths, 'running', timeout);
|
||||
|
||||
// Enforce max tools limit
|
||||
if (this._activeTools.size >= MAX_ACTIVE_TOOLS) {
|
||||
// Remove oldest tool (O(n) min-scan instead of O(n log n) sort)
|
||||
let oldestKey: string | undefined;
|
||||
let oldestTime = Infinity;
|
||||
for (const [key, entry] of this._activeTools) {
|
||||
if (entry.startedAt < oldestTime) {
|
||||
oldestTime = entry.startedAt;
|
||||
oldestKey = key;
|
||||
}
|
||||
|
||||
this._activeTools.set(tool.id, tool);
|
||||
this._lastToolId = tool.id;
|
||||
|
||||
this.emit('toolStart', tool);
|
||||
this.scheduleUpdate();
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// Check for tool completion
|
||||
if (TOOL_COMPLETION_PATTERN.test(cleanLine) && this._lastToolId) {
|
||||
const tool = this._activeTools.get(this._lastToolId);
|
||||
if (tool && tool.status === 'running') {
|
||||
tool.status = 'completed';
|
||||
this.emit('toolEnd', tool);
|
||||
this.scheduleUpdate();
|
||||
|
||||
// Remove completed tool after a short delay to allow UI to show completion
|
||||
this.cleanup.setTimeout(
|
||||
() => {
|
||||
if (this._destroyed) return;
|
||||
this._activeTools.delete(tool.id);
|
||||
this.scheduleUpdate();
|
||||
},
|
||||
2000,
|
||||
{ description: 'auto-remove completed tool' }
|
||||
);
|
||||
}
|
||||
this._lastToolId = null;
|
||||
return;
|
||||
}
|
||||
|
||||
// Fallback: Check for command suggestions in plain text (e.g., "tail -f /tmp/file.log")
|
||||
const textCmdMatch = cleanLine.match(TEXT_COMMAND_PATTERN);
|
||||
if (textCmdMatch) {
|
||||
const filePath = textCmdMatch[2];
|
||||
|
||||
// Create a suggestion tool (marked as 'suggestion' status)
|
||||
const tool: ActiveBashTool = {
|
||||
id: uuidv4(),
|
||||
command: cleanLine.trim(),
|
||||
filePaths: [filePath],
|
||||
timeout: undefined,
|
||||
startedAt: Date.now(),
|
||||
status: 'running', // Shows as clickable
|
||||
sessionId: this._sessionId,
|
||||
};
|
||||
|
||||
// Don't add if file path already tracked (cross-pattern dedup)
|
||||
if (this.isFilePathTracked(filePath)) {
|
||||
return;
|
||||
if (oldestKey) {
|
||||
this._activeTools.delete(oldestKey);
|
||||
}
|
||||
}
|
||||
|
||||
this._activeTools.set(tool.id, tool);
|
||||
this._lastToolId = tool.id;
|
||||
|
||||
this.emit('toolStart', tool);
|
||||
this.scheduleUpdate();
|
||||
|
||||
// Auto-remove suggestions after 30 seconds
|
||||
this.cleanup.setTimeout(
|
||||
() => {
|
||||
if (this._destroyed) return;
|
||||
this._activeTools.delete(tool.id);
|
||||
this.scheduleUpdate();
|
||||
},
|
||||
30000,
|
||||
{ description: 'auto-remove suggestion tool' }
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
// Last fallback: Check for log file paths mentioned anywhere in the line
|
||||
return true;
|
||||
}
|
||||
|
||||
private _handleToolCompletion(cleanLine: string): boolean {
|
||||
if (!TOOL_COMPLETION_PATTERN.test(cleanLine) || !this._lastToolId) return false;
|
||||
|
||||
const tool = this._activeTools.get(this._lastToolId);
|
||||
if (tool && tool.status === 'running') {
|
||||
tool.status = 'completed';
|
||||
this.emit('toolEnd', tool);
|
||||
this.scheduleUpdate();
|
||||
|
||||
this._scheduleAutoRemove(tool.id, 2000, 'auto-remove completed tool');
|
||||
}
|
||||
this._lastToolId = null;
|
||||
return true;
|
||||
}
|
||||
|
||||
private _handleTextCommand(cleanLine: string): boolean {
|
||||
const textCmdMatch = cleanLine.match(TEXT_COMMAND_PATTERN);
|
||||
if (!textCmdMatch) return false;
|
||||
|
||||
const filePath = textCmdMatch[2];
|
||||
|
||||
// Don't add if file path already tracked (cross-pattern dedup)
|
||||
if (this.isFilePathTracked(filePath)) return true;
|
||||
|
||||
const tool = this._createActiveTool(cleanLine.trim(), [filePath], 'running');
|
||||
|
||||
this._activeTools.set(tool.id, tool);
|
||||
this.emit('toolStart', tool);
|
||||
this.scheduleUpdate();
|
||||
|
||||
// Auto-remove suggestions after 30 seconds
|
||||
this._scheduleAutoRemove(tool.id, 30000, 'auto-remove suggestion tool');
|
||||
return true;
|
||||
}
|
||||
|
||||
private _handleLogFileMention(cleanLine: string): void {
|
||||
LOG_FILE_MENTION_PATTERN.lastIndex = 0;
|
||||
let logMatch;
|
||||
while ((logMatch = LOG_FILE_MENTION_PATTERN.exec(cleanLine)) !== null) {
|
||||
@@ -588,33 +566,46 @@ export class BashToolParser extends EventEmitter<BashToolParserEvents> {
|
||||
// Skip if file path already tracked (cross-pattern dedup)
|
||||
if (this.isFilePathTracked(filePath)) continue;
|
||||
|
||||
const tool: ActiveBashTool = {
|
||||
id: uuidv4(),
|
||||
command: `View: ${filePath}`,
|
||||
filePaths: [filePath],
|
||||
timeout: undefined,
|
||||
startedAt: Date.now(),
|
||||
status: 'running',
|
||||
sessionId: this._sessionId,
|
||||
};
|
||||
const tool = this._createActiveTool(`View: ${filePath}`, [filePath], 'running');
|
||||
|
||||
this._activeTools.set(tool.id, tool);
|
||||
this.emit('toolStart', tool);
|
||||
this.scheduleUpdate();
|
||||
|
||||
// Auto-remove after 60 seconds
|
||||
this.cleanup.setTimeout(
|
||||
() => {
|
||||
if (this._destroyed) return;
|
||||
this._activeTools.delete(tool.id);
|
||||
this.scheduleUpdate();
|
||||
},
|
||||
60000,
|
||||
{ description: 'auto-remove log file tool' }
|
||||
);
|
||||
this._scheduleAutoRemove(tool.id, 60000, 'auto-remove log file tool');
|
||||
}
|
||||
}
|
||||
|
||||
private _createActiveTool(
|
||||
command: string,
|
||||
filePaths: string[],
|
||||
status: ActiveBashTool['status'],
|
||||
timeout?: string
|
||||
): ActiveBashTool {
|
||||
return {
|
||||
id: uuidv4(),
|
||||
command,
|
||||
filePaths,
|
||||
timeout,
|
||||
startedAt: Date.now(),
|
||||
status,
|
||||
sessionId: this._sessionId,
|
||||
};
|
||||
}
|
||||
|
||||
private _scheduleAutoRemove(toolId: string, delayMs: number, description: string): void {
|
||||
this.cleanup.setTimeout(
|
||||
() => {
|
||||
if (this._destroyed) return;
|
||||
this._activeTools.delete(toolId);
|
||||
this.scheduleUpdate();
|
||||
},
|
||||
delayMs,
|
||||
{ description }
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a command is a file-viewing command worth tracking.
|
||||
*/
|
||||
|
||||
@@ -483,19 +483,29 @@ program
|
||||
program
|
||||
.command('web')
|
||||
.description('Start the web interface')
|
||||
.option('-p, --port <port>', 'Port to listen on', '3000')
|
||||
.option('-H, --host <host>', 'Host to bind to', process.env.CODEMAN_HOST || '127.0.0.1')
|
||||
.option('-p, --port <port>', 'Port to listen on (env: CODEMAN_PORT)', process.env.CODEMAN_PORT || '3000')
|
||||
.option('--https', 'Enable HTTPS with self-signed certificate (only needed for remote access, not localhost)')
|
||||
.option('--title-hostname <hostname>', 'Override the hostname shown in the browser title')
|
||||
.option(
|
||||
'--allow-unauthenticated-network',
|
||||
'Allow non-loopback web access without CODEMAN_PASSWORD (dangerous; terminal control is exposed)'
|
||||
)
|
||||
.action(async (options) => {
|
||||
const { startWebServer } = await import('./web/server.js');
|
||||
const host = options.host;
|
||||
const port = parseInt(options.port, 10);
|
||||
const https = !!options.https;
|
||||
const titleHostname = options.titleHostname;
|
||||
const allowUnauthenticatedNetwork = !!options.allowUnauthenticatedNetwork;
|
||||
const protocol = https ? 'https' : 'http';
|
||||
const displayHost = host === '0.0.0.0' ? 'localhost' : host;
|
||||
|
||||
console.log(chalk.cyan(`Starting Codeman web interface on port ${port}${https ? ' (HTTPS)' : ''}...`));
|
||||
console.log(chalk.cyan(`Starting Codeman web interface on ${displayHost}:${port}${https ? ' (HTTPS)' : ''}...`));
|
||||
|
||||
try {
|
||||
const server = await startWebServer(port, https);
|
||||
console.log(chalk.green(`\n✓ Web interface running at ${protocol}://localhost:${port}`));
|
||||
const server = await startWebServer(port, https, false, host, titleHostname, allowUnauthenticatedNetwork);
|
||||
console.log(chalk.green(`\n✓ Web interface running at ${protocol}://${displayHost}:${port}`));
|
||||
if (https) {
|
||||
console.log(chalk.yellow(' Note: Accept the self-signed certificate in your browser on first visit'));
|
||||
}
|
||||
|
||||
@@ -22,14 +22,16 @@
|
||||
* Maximum terminal buffer size in characters.
|
||||
* Contains raw terminal output with ANSI escape sequences.
|
||||
* Reduced from 5MB to 2MB for better render performance.
|
||||
* Override: CODEMAN_MAX_TERMINAL_BUFFER (bytes)
|
||||
*/
|
||||
export const MAX_TERMINAL_BUFFER_SIZE = 2 * 1024 * 1024; // 2MB
|
||||
export const MAX_TERMINAL_BUFFER_SIZE = parseInt(process.env.CODEMAN_MAX_TERMINAL_BUFFER || '') || 2 * 1024 * 1024;
|
||||
|
||||
/**
|
||||
* Size to trim terminal buffer to when max is exceeded.
|
||||
* Keeps the most recent portion to preserve context.
|
||||
* Override: CODEMAN_TRIM_TERMINAL_TO (bytes)
|
||||
*/
|
||||
export const TRIM_TERMINAL_TO = 1.5 * 1024 * 1024; // 1.5MB
|
||||
export const TRIM_TERMINAL_TO = parseInt(process.env.CODEMAN_TRIM_TERMINAL_TO || '') || 1.5 * 1024 * 1024;
|
||||
|
||||
// ============================================================================
|
||||
// Text Output Buffer Limits
|
||||
@@ -38,13 +40,15 @@ export const TRIM_TERMINAL_TO = 1.5 * 1024 * 1024; // 1.5MB
|
||||
/**
|
||||
* Maximum text output buffer size in characters.
|
||||
* Contains ANSI-stripped text for search and analysis.
|
||||
* Override: CODEMAN_MAX_TEXT_OUTPUT (bytes)
|
||||
*/
|
||||
export const MAX_TEXT_OUTPUT_SIZE = 1 * 1024 * 1024; // 1MB
|
||||
export const MAX_TEXT_OUTPUT_SIZE = parseInt(process.env.CODEMAN_MAX_TEXT_OUTPUT || '') || 1 * 1024 * 1024;
|
||||
|
||||
/**
|
||||
* Size to trim text output buffer to when max is exceeded.
|
||||
* Override: CODEMAN_TRIM_TEXT_TO (bytes)
|
||||
*/
|
||||
export const TRIM_TEXT_TO = 768 * 1024; // 768KB
|
||||
export const TRIM_TEXT_TO = parseInt(process.env.CODEMAN_TRIM_TEXT_TO || '') || 768 * 1024;
|
||||
|
||||
// ============================================================================
|
||||
// Message Buffer Limits
|
||||
@@ -53,8 +57,9 @@ export const TRIM_TEXT_TO = 768 * 1024; // 768KB
|
||||
/**
|
||||
* Maximum number of Claude JSON messages to keep in memory per session.
|
||||
* Older messages are discarded when limit is exceeded.
|
||||
* Override: CODEMAN_MAX_MESSAGES (count)
|
||||
*/
|
||||
export const MAX_MESSAGES = 1000;
|
||||
export const MAX_MESSAGES = parseInt(process.env.CODEMAN_MAX_MESSAGES || '') || 1000;
|
||||
|
||||
// ============================================================================
|
||||
// Line Buffer Limits
|
||||
|
||||
@@ -0,0 +1,66 @@
|
||||
/**
|
||||
* @fileoverview Per-instance isolation: data directory + tmux socket.
|
||||
*
|
||||
* Codeman keeps all runtime state under `~/.codeman` and runs its tmux sessions
|
||||
* on a dedicated socket (`tmux -L codeman`). Both are PROCESS-WIDE and SHARED by
|
||||
* every Codeman instance on the machine — so a second instance pointed at the
|
||||
* same socket will discover and attach to the first instance's live sessions,
|
||||
* and two instances sharing `~/.codeman/state.json` will clobber each other.
|
||||
*
|
||||
* To let a beta build coexist with a production one, this module derives both
|
||||
* the data dir and the tmux socket from a single "instance" name:
|
||||
* - default (unset/empty) → `~/.codeman` + `tmux -L codeman` (prod layout)
|
||||
* - `CODEMAN_INSTANCE=beta` → `~/.codeman-beta` + `tmux -L codeman-beta`
|
||||
* - `CODEMAN_INSTANCE=foo` → `~/.codeman-foo` + `tmux -L codeman-foo`
|
||||
*
|
||||
* The DEFAULT is the production layout so this is safe to ship to master: an
|
||||
* existing install keeps reading `~/.codeman`. To run a beta ALONGSIDE prod,
|
||||
* launch it with `CODEMAN_INSTANCE=beta` (and a distinct port, see below) —
|
||||
* `scripts/run-beta.sh` does both. The port is unrelated to the instance and is
|
||||
* set separately via `--port` / `CODEMAN_PORT` (see `src/cli.ts`).
|
||||
*
|
||||
* Individual overrides still win: `CODEMAN_DATA_DIR` (absolute data dir) and
|
||||
* `CODEMAN_TMUX_SOCKET` (socket name, validated in tmux-manager).
|
||||
*/
|
||||
|
||||
import { homedir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { mkdirSync } from 'node:fs';
|
||||
|
||||
/**
|
||||
* Instance name. Empty string (the default) = production layout (`~/.codeman`,
|
||||
* `-L codeman`), so this is safe on master and existing installs are untouched.
|
||||
* Set `CODEMAN_INSTANCE=beta` (e.g. via `scripts/run-beta.sh`) to run an
|
||||
* isolated beta alongside prod.
|
||||
*/
|
||||
export const CODEMAN_INSTANCE = process.env.CODEMAN_INSTANCE ?? '';
|
||||
|
||||
const INSTANCE_SUFFIX = CODEMAN_INSTANCE ? `-${CODEMAN_INSTANCE}` : '';
|
||||
|
||||
/** Default tmux socket for this instance. `CODEMAN_TMUX_SOCKET` still overrides. */
|
||||
export const DEFAULT_TMUX_SOCKET = `codeman${INSTANCE_SUFFIX}`;
|
||||
|
||||
let _ensured = false;
|
||||
|
||||
/**
|
||||
* Absolute path to this instance's data directory (created on first use). All
|
||||
* persisted state (`state.json`, `mux-sessions.json`, settings, push keys,
|
||||
* lifecycle log, screenshots, certs, …) lives here.
|
||||
*/
|
||||
export function getDataDir(): string {
|
||||
const dir = process.env.CODEMAN_DATA_DIR || join(homedir(), `.codeman${INSTANCE_SUFFIX}`);
|
||||
if (!_ensured) {
|
||||
try {
|
||||
mkdirSync(dir, { recursive: true });
|
||||
_ensured = true;
|
||||
} catch {
|
||||
/* best-effort; individual writers also mkdir as needed */
|
||||
}
|
||||
}
|
||||
return dir;
|
||||
}
|
||||
|
||||
/** Join one or more segments onto this instance's data directory. */
|
||||
export function dataPath(...segments: string[]): string {
|
||||
return join(getDataDir(), ...segments);
|
||||
}
|
||||
@@ -16,6 +16,7 @@ import { existsSync, statSync, realpathSync } from 'node:fs';
|
||||
import { resolve, relative, isAbsolute } from 'node:path';
|
||||
import { homedir } from 'node:os';
|
||||
import { EventEmitter } from 'node:events';
|
||||
import { getErrorMessage } from './types.js';
|
||||
import { CLEANUP_CHECK_INTERVAL_MS, INACTIVITY_TIMEOUT_MS } from './config/server-timing.js';
|
||||
|
||||
// ========== Configuration Constants ==========
|
||||
@@ -47,7 +48,7 @@ const STREAM_INACTIVITY_TIMEOUT_MS = INACTIVITY_TIMEOUT_MS;
|
||||
/**
|
||||
* Represents an active file stream.
|
||||
*/
|
||||
export interface FileStream {
|
||||
interface FileStream {
|
||||
/** Unique stream identifier */
|
||||
id: string;
|
||||
/** Session this stream belongs to */
|
||||
@@ -73,7 +74,7 @@ export interface FileStream {
|
||||
/**
|
||||
* Options for creating a file stream.
|
||||
*/
|
||||
export interface CreateStreamOptions {
|
||||
interface CreateStreamOptions {
|
||||
/** Session ID requesting the stream */
|
||||
sessionId: string;
|
||||
/** Path to the file to stream */
|
||||
@@ -93,7 +94,7 @@ export interface CreateStreamOptions {
|
||||
/**
|
||||
* Result of creating a stream.
|
||||
*/
|
||||
export interface CreateStreamResult {
|
||||
interface CreateStreamResult {
|
||||
success: boolean;
|
||||
streamId?: string;
|
||||
error?: string;
|
||||
@@ -172,10 +173,7 @@ export class FileStreamManager extends EventEmitter {
|
||||
}
|
||||
} catch (err) {
|
||||
const errorCode = err instanceof Error && 'code' in err ? (err as NodeJS.ErrnoException).code : 'UNKNOWN';
|
||||
console.warn(
|
||||
`[FileStreamManager] Failed to stat file "${absolutePath}" (${errorCode}):`,
|
||||
err instanceof Error ? err.message : String(err)
|
||||
);
|
||||
console.warn(`[FileStreamManager] Failed to stat file "${absolutePath}" (${errorCode}):`, getErrorMessage(err));
|
||||
return { success: false, error: 'File not found or not accessible' };
|
||||
}
|
||||
|
||||
|
||||
@@ -84,6 +84,42 @@ export function generateHooksConfig(): { hooks: Record<string, unknown[]> } {
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Remove a subset of env keys from .claude/settings.local.json.env if present.
|
||||
* Used during the disk→tmux-setenv migration: when the caller is actively setting
|
||||
* a fresh value for a Codeman-managed key, any stale disk entry for THAT KEY is
|
||||
* superseded and should be removed. Keys NOT in `keysToRemove` are left alone
|
||||
* (they may be user-managed). No-op if the file/keys don't exist.
|
||||
*/
|
||||
export async function stripCaseEnvKeys(casePath: string, keysToRemove: readonly string[]): Promise<void> {
|
||||
if (keysToRemove.length === 0) return;
|
||||
|
||||
const settingsPath = join(casePath, '.claude', 'settings.local.json');
|
||||
if (!existsSync(settingsPath)) return;
|
||||
|
||||
let existing: Record<string, unknown>;
|
||||
try {
|
||||
existing = JSON.parse(await readFile(settingsPath, 'utf-8'));
|
||||
} catch {
|
||||
return; // Malformed — don't rewrite it
|
||||
}
|
||||
|
||||
const env = existing.env as Record<string, string> | undefined;
|
||||
if (!env) return;
|
||||
|
||||
let changed = false;
|
||||
for (const key of keysToRemove) {
|
||||
if (key in env) {
|
||||
delete env[key];
|
||||
changed = true;
|
||||
}
|
||||
}
|
||||
if (!changed) return;
|
||||
|
||||
existing.env = env;
|
||||
await writeFile(settingsPath, JSON.stringify(existing, null, 2) + '\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Updates env vars in .claude/settings.local.json for the given case path.
|
||||
* Merges with existing env field; removes vars set to empty string.
|
||||
@@ -116,6 +152,34 @@ export async function updateCaseEnvVars(casePath: string, envVars: Record<string
|
||||
await writeFile(settingsPath, JSON.stringify(existing, null, 2) + '\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Updates the `model` field in .claude/settings.local.json for the given case path.
|
||||
* Pass a non-empty string to set, or empty/null to remove.
|
||||
*/
|
||||
export async function updateCaseModel(casePath: string, model: string | null): Promise<void> {
|
||||
const claudeDir = join(casePath, '.claude');
|
||||
if (!existsSync(claudeDir)) {
|
||||
await mkdir(claudeDir, { recursive: true });
|
||||
}
|
||||
|
||||
const settingsPath = join(claudeDir, 'settings.local.json');
|
||||
let existing: Record<string, unknown> = {};
|
||||
|
||||
try {
|
||||
existing = JSON.parse(await readFile(settingsPath, 'utf-8'));
|
||||
} catch {
|
||||
existing = {};
|
||||
}
|
||||
|
||||
if (model) {
|
||||
existing.model = model;
|
||||
} else {
|
||||
delete existing.model;
|
||||
}
|
||||
|
||||
await writeFile(settingsPath, JSON.stringify(existing, null, 2) + '\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Writes hooks config to .claude/settings.local.json in the given case path.
|
||||
* Merges with existing file content, only touching the `hooks` key.
|
||||
|
||||
@@ -17,11 +17,6 @@ import { KeyedDebouncer } from './utils/index.js';
|
||||
|
||||
// ========== Types ==========
|
||||
|
||||
export interface ImageWatcherEvents {
|
||||
'image:detected': (event: ImageDetectedEvent) => void;
|
||||
'image:error': (error: Error, sessionId?: string) => void;
|
||||
}
|
||||
|
||||
// ========== Constants ==========
|
||||
|
||||
/** Supported image file extensions (lowercase) */
|
||||
|
||||
@@ -14,6 +14,7 @@ import type {
|
||||
ClaudeMode,
|
||||
SessionMode,
|
||||
OpenCodeConfig,
|
||||
EffortLevel,
|
||||
} from './types.js';
|
||||
|
||||
/**
|
||||
@@ -63,6 +64,10 @@ export interface CreateSessionOptions {
|
||||
openCodeConfig?: OpenCodeConfig;
|
||||
/** When restoring after reboot, resume a previous Claude conversation by its session ID */
|
||||
resumeSessionId?: string;
|
||||
/** Extra env vars exported before launching the CLI (e.g., CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS). Ephemeral — not written to disk. */
|
||||
envOverrides?: Record<string, string>;
|
||||
/** Claude CLI effort level, injected as a `--settings` soft default (overridable via /effort in-session) */
|
||||
effort?: EffortLevel;
|
||||
}
|
||||
|
||||
/** Options for respawning a dead pane. */
|
||||
@@ -77,6 +82,10 @@ export interface RespawnPaneOptions {
|
||||
openCodeConfig?: OpenCodeConfig;
|
||||
/** Resume a previous Claude conversation when respawning */
|
||||
resumeSessionId?: string;
|
||||
/** Extra env vars exported before launching the CLI (preserved across respawns). */
|
||||
envOverrides?: Record<string, string>;
|
||||
/** Claude CLI effort level (preserved across respawns, injected via `--settings`) */
|
||||
effort?: EffortLevel;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -94,6 +103,9 @@ export interface TerminalMultiplexer extends EventEmitter {
|
||||
/** Which backend this instance uses */
|
||||
readonly backend: 'tmux';
|
||||
|
||||
/** The dedicated tmux socket name all sessions live on (e.g. "codeman"). */
|
||||
readonly muxSocket: string;
|
||||
|
||||
// ========== Lifecycle ==========
|
||||
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,978 @@
|
||||
/**
|
||||
* @fileoverview Orchestrator Loop — phased plan execution with team agents.
|
||||
*
|
||||
* State machine that generates plans from user goals, executes them
|
||||
* phase-by-phase with verification gates, and adapts on failure.
|
||||
*
|
||||
* States: idle → planning → approval → executing → verifying → (replanning) → completed/failed
|
||||
*
|
||||
* Key exports:
|
||||
* - `OrchestratorLoop` class — main engine, extends EventEmitter
|
||||
* - `OrchestratorLoopEvents` interface — typed event map
|
||||
*
|
||||
* Lifecycle: `start(goal)` → plan → approve → execute phases → verify → complete
|
||||
*
|
||||
* @dependencies orchestrator-planner (plan generation), orchestrator-verifier (phase verification),
|
||||
* session-manager (sessions), task-queue (task execution), state-store (persistence),
|
||||
* prompts/orchestrator (prompt templates)
|
||||
* @consumedby web/server (orchestrator routes, SSE)
|
||||
* @emits stateChanged, planReady, phaseStarted, phaseCompleted, phaseFailed,
|
||||
* taskAssigned, taskCompleted, taskFailed, verificationResult, completed, error
|
||||
* @persistence Orchestrator state saved to `~/.codeman/state.json` (orchestrator key)
|
||||
*
|
||||
* @module orchestrator-loop
|
||||
*/
|
||||
|
||||
import { EventEmitter } from 'node:events';
|
||||
import { getSessionManager, SessionManager } from './session-manager.js';
|
||||
import { getTaskQueue, TaskQueue } from './task-queue.js';
|
||||
import { getStore, StateStore } from './state-store.js';
|
||||
import { OrchestratorPlanner } from './orchestrator-planner.js';
|
||||
import { OrchestratorVerifier } from './orchestrator-verifier.js';
|
||||
import { PHASE_EXECUTION_PROMPT, REPLAN_PROMPT, SINGLE_TASK_PROMPT, TEAM_LEAD_PROMPT } from './prompts/index.js';
|
||||
import type { TerminalMultiplexer } from './mux-interface.js';
|
||||
import type { CreateTaskOptions } from './task.js';
|
||||
import {
|
||||
type OrchestratorState,
|
||||
type OrchestratorPlan,
|
||||
type OrchestratorPhase,
|
||||
type OrchestratorTask,
|
||||
type OrchestratorConfig,
|
||||
type OrchestratorStats,
|
||||
type OrchestratorPersistState,
|
||||
type VerificationResult,
|
||||
DEFAULT_ORCHESTRATOR_CONFIG,
|
||||
createInitialOrchestratorStats,
|
||||
getErrorMessage,
|
||||
} from './types.js';
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Constants
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
/** Poll interval for checking task completion within a phase (2 seconds) */
|
||||
const PHASE_POLL_INTERVAL_MS = 2000;
|
||||
|
||||
/** Delay between phase completion and verification (1 second) */
|
||||
const POST_PHASE_DELAY_MS = 1000;
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Events
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// OrchestratorLoop
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
export class OrchestratorLoop extends EventEmitter {
|
||||
private _state: OrchestratorState = 'idle';
|
||||
private plan: OrchestratorPlan | null = null;
|
||||
private currentPhaseIndex = 0;
|
||||
private config: OrchestratorConfig;
|
||||
private stats: OrchestratorStats;
|
||||
private startedAt: number | null = null;
|
||||
private completedAt: number | null = null;
|
||||
|
||||
private workingDir: string;
|
||||
private planner: OrchestratorPlanner;
|
||||
private verifier: OrchestratorVerifier;
|
||||
private sessionManager: SessionManager;
|
||||
private taskQueue: TaskQueue;
|
||||
private store: StateStore;
|
||||
|
||||
/** State before pause (to resume to correct state) */
|
||||
private pausedState: OrchestratorState | null = null;
|
||||
|
||||
/** Phase poll timer for checking task completion */
|
||||
private phasePollTimer: NodeJS.Timeout | null = null;
|
||||
|
||||
/** Phase-level timeout timer */
|
||||
private phaseTimeoutTimer: NodeJS.Timeout | null = null;
|
||||
|
||||
/** Post-phase delay timer before verification */
|
||||
private postPhaseTimer: NodeJS.Timeout | null = null;
|
||||
|
||||
/** Session completion listener (bound for cleanup) */
|
||||
private sessionCompletionListener: ((sessionId: string, phrase: string) => void) | null = null;
|
||||
|
||||
/** Active sessions assigned to current phase */
|
||||
private phaseSessionIds: Set<string> = new Set();
|
||||
|
||||
constructor(mux: TerminalMultiplexer, workingDir: string, config?: Partial<OrchestratorConfig>) {
|
||||
super();
|
||||
this.workingDir = workingDir;
|
||||
this.config = { ...DEFAULT_ORCHESTRATOR_CONFIG, ...config };
|
||||
this.stats = createInitialOrchestratorStats();
|
||||
this.sessionManager = getSessionManager();
|
||||
this.taskQueue = getTaskQueue();
|
||||
this.store = getStore();
|
||||
this.planner = new OrchestratorPlanner(mux, workingDir, this.config);
|
||||
this.verifier = new OrchestratorVerifier(this.config);
|
||||
|
||||
// Restore state if crashed while running
|
||||
this.restore();
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Public API — Lifecycle
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
/** Start orchestration with a goal. Transitions: idle → planning */
|
||||
async start(goal: string): Promise<void> {
|
||||
if (this._state !== 'idle' && this._state !== 'failed' && this._state !== 'completed') {
|
||||
throw new Error(`Cannot start from state "${this._state}"`);
|
||||
}
|
||||
|
||||
this.reset();
|
||||
this.startedAt = Date.now();
|
||||
this.setState('planning');
|
||||
|
||||
try {
|
||||
const plan = await this.planner.generatePlan(goal, (phase, detail) => {
|
||||
this.emit('planProgress', phase, detail);
|
||||
});
|
||||
|
||||
if (this.currentState() !== 'planning') {
|
||||
// Cancelled during planning
|
||||
return;
|
||||
}
|
||||
|
||||
this.plan = plan;
|
||||
this.persist();
|
||||
|
||||
if (this.config.autoApprove) {
|
||||
this.setState('executing');
|
||||
await this.executeCurrentPhase();
|
||||
} else {
|
||||
this.setState('approval');
|
||||
this.emit('planReady', plan);
|
||||
}
|
||||
} catch (err) {
|
||||
this.handleError(err);
|
||||
}
|
||||
}
|
||||
|
||||
/** Approve the generated plan. Transitions: approval → executing */
|
||||
async approve(): Promise<void> {
|
||||
this.requireState('approval');
|
||||
if (!this.plan) {
|
||||
throw new Error('No plan to approve');
|
||||
}
|
||||
|
||||
this.setState('executing');
|
||||
await this.executeCurrentPhase();
|
||||
}
|
||||
|
||||
/** Reject plan with feedback. Transitions: approval → planning (regenerate) */
|
||||
async reject(feedback: string): Promise<void> {
|
||||
this.requireState('approval');
|
||||
if (!this.plan) {
|
||||
throw new Error('No plan to reject');
|
||||
}
|
||||
|
||||
const goal = this.plan.goal + '\n\nFeedback on previous plan: ' + feedback;
|
||||
this.plan = null;
|
||||
this.setState('planning');
|
||||
|
||||
try {
|
||||
const plan = await this.planner.generatePlan(goal);
|
||||
|
||||
if ((this._state as OrchestratorState) !== 'planning') return;
|
||||
|
||||
this.plan = plan;
|
||||
this.persist();
|
||||
this.setState('approval');
|
||||
this.emit('planReady', plan);
|
||||
} catch (err) {
|
||||
this.handleError(err);
|
||||
}
|
||||
}
|
||||
|
||||
/** Pause execution. Saves current state. */
|
||||
pause(): void {
|
||||
if (this._state === 'idle' || this._state === 'paused' || this._state === 'completed' || this._state === 'failed') {
|
||||
return;
|
||||
}
|
||||
this.pausedState = this._state;
|
||||
this.clearPhasePoll();
|
||||
this.cleanupTaskHandlers();
|
||||
this.setState('paused');
|
||||
}
|
||||
|
||||
/** Resume from pause. */
|
||||
async resume(): Promise<void> {
|
||||
if (this._state !== 'paused' || !this.pausedState) {
|
||||
throw new Error('Not paused');
|
||||
}
|
||||
|
||||
const resumeTo = this.pausedState;
|
||||
this.pausedState = null;
|
||||
this.setState(resumeTo);
|
||||
|
||||
// Re-enter the appropriate phase of execution
|
||||
if (resumeTo === 'executing') {
|
||||
await this.executeCurrentPhase();
|
||||
} else if (resumeTo === 'verifying') {
|
||||
await this.verifyCurrentPhase();
|
||||
}
|
||||
}
|
||||
|
||||
/** Stop everything and clean up. */
|
||||
async stop(): Promise<void> {
|
||||
this.clearPhasePoll();
|
||||
this.cleanupTaskHandlers();
|
||||
await this.planner.cancel();
|
||||
this.setState('idle');
|
||||
this.store.clearOrchestratorState();
|
||||
}
|
||||
|
||||
/** Skip a specific phase. */
|
||||
async skipPhase(phaseId: string): Promise<void> {
|
||||
if (!this.plan) return;
|
||||
|
||||
const phase = this.plan.phases.find((p) => p.id === phaseId);
|
||||
if (!phase) throw new Error(`Phase "${phaseId}" not found`);
|
||||
|
||||
phase.status = 'skipped';
|
||||
phase.completedAt = Date.now();
|
||||
this.persist();
|
||||
|
||||
// If this is the current phase, advance
|
||||
if (this.plan.phases[this.currentPhaseIndex]?.id === phaseId) {
|
||||
await this.advanceToNextPhase();
|
||||
}
|
||||
}
|
||||
|
||||
/** Retry a failed phase. */
|
||||
async retryPhase(phaseId: string): Promise<void> {
|
||||
if (!this.plan) return;
|
||||
if (this._state !== 'executing' && this._state !== 'failed') {
|
||||
throw new Error(`Cannot retry from state "${this._state}"`);
|
||||
}
|
||||
|
||||
const phaseIndex = this.plan.phases.findIndex((p) => p.id === phaseId);
|
||||
if (phaseIndex === -1) throw new Error(`Phase "${phaseId}" not found`);
|
||||
|
||||
const phase = this.plan.phases[phaseIndex];
|
||||
phase.status = 'pending';
|
||||
phase.attempts = 0;
|
||||
for (const task of phase.tasks) {
|
||||
task.status = 'pending';
|
||||
task.error = null;
|
||||
task.assignedSessionId = null;
|
||||
task.queueTaskId = null;
|
||||
}
|
||||
|
||||
this.currentPhaseIndex = phaseIndex;
|
||||
this.setState('executing');
|
||||
await this.executeCurrentPhase();
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Public API — Getters
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
get state(): OrchestratorState {
|
||||
return this._state;
|
||||
}
|
||||
|
||||
getPlan(): OrchestratorPlan | null {
|
||||
return this.plan;
|
||||
}
|
||||
|
||||
getCurrentPhase(): OrchestratorPhase | null {
|
||||
if (!this.plan) return null;
|
||||
return this.plan.phases[this.currentPhaseIndex] ?? null;
|
||||
}
|
||||
|
||||
getStats(): OrchestratorStats {
|
||||
return { ...this.stats };
|
||||
}
|
||||
|
||||
getStatus(): OrchestratorPersistState {
|
||||
return {
|
||||
state: this._state,
|
||||
plan: this.plan,
|
||||
currentPhaseIndex: this.currentPhaseIndex,
|
||||
startedAt: this.startedAt,
|
||||
completedAt: this.completedAt,
|
||||
config: this.config,
|
||||
stats: this.stats,
|
||||
};
|
||||
}
|
||||
|
||||
isRunning(): boolean {
|
||||
return this._state !== 'idle' && this._state !== 'completed' && this._state !== 'failed';
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Internal — Phase Execution
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
private async executeCurrentPhase(): Promise<void> {
|
||||
if (!this.plan || this._state !== 'executing') return;
|
||||
|
||||
const phase = this.plan.phases[this.currentPhaseIndex];
|
||||
if (!phase) {
|
||||
// All phases done
|
||||
await this.handleCompletion();
|
||||
return;
|
||||
}
|
||||
|
||||
// Skip already completed/skipped phases
|
||||
if (phase.status === 'passed' || phase.status === 'skipped') {
|
||||
await this.advanceToNextPhase();
|
||||
return;
|
||||
}
|
||||
|
||||
phase.status = 'executing';
|
||||
phase.startedAt = Date.now();
|
||||
phase.attempts++;
|
||||
this.persist();
|
||||
this.emit('phaseStarted', phase);
|
||||
|
||||
try {
|
||||
await this.assignPhaseTasks(phase);
|
||||
this.startPhasePoll(phase);
|
||||
} catch (err) {
|
||||
this.handlePhaseError(phase, getErrorMessage(err));
|
||||
}
|
||||
}
|
||||
|
||||
private async assignPhaseTasks(phase: OrchestratorPhase): Promise<void> {
|
||||
// For team strategy, send a single comprehensive prompt to a lead session
|
||||
if (phase.teamStrategy.type === 'team') {
|
||||
await this.assignTeamPhase(phase);
|
||||
return;
|
||||
}
|
||||
|
||||
// For single/parallel strategy, add individual tasks to TaskQueue
|
||||
for (const task of phase.tasks) {
|
||||
if (task.status !== 'pending') continue;
|
||||
|
||||
const prompt = this.buildTaskPrompt(task, phase);
|
||||
const taskOptions: CreateTaskOptions = {
|
||||
prompt,
|
||||
workingDir: this.workingDir,
|
||||
priority: 100 - phase.order, // Earlier phases get higher priority
|
||||
completionPhrase: task.completionPhrase,
|
||||
timeoutMs: Math.min(task.timeoutMs, this.config.phaseTimeoutMs),
|
||||
};
|
||||
|
||||
const queueTask = this.taskQueue.addTask(taskOptions);
|
||||
task.queueTaskId = queueTask.id;
|
||||
task.status = 'running';
|
||||
}
|
||||
|
||||
this.persist();
|
||||
this.setupTaskHandlers();
|
||||
|
||||
// Manually assign tasks to idle sessions
|
||||
await this.assignQueuedTasksToSessions();
|
||||
}
|
||||
|
||||
private async assignTeamPhase(phase: OrchestratorPhase): Promise<void> {
|
||||
const teamConfig = phase.teamStrategy.type === 'team' ? phase.teamStrategy.config : null;
|
||||
if (!teamConfig) return;
|
||||
|
||||
// Find or use an idle session
|
||||
const sessions = this.sessionManager.getIdleSessions();
|
||||
if (sessions.length === 0) {
|
||||
throw new Error('No idle sessions available for team phase execution');
|
||||
}
|
||||
|
||||
const session = sessions[0];
|
||||
this.phaseSessionIds.add(session.id);
|
||||
|
||||
// Mark all tasks as running under this session
|
||||
for (const task of phase.tasks) {
|
||||
task.status = 'running';
|
||||
task.assignedSessionId = session.id;
|
||||
}
|
||||
|
||||
// Build and send the team lead prompt
|
||||
const prompt = TEAM_LEAD_PROMPT.replace('{PHASE_NAME}', phase.name)
|
||||
.replace('{TASK_LIST}', phase.tasks.map((t, i) => `${i + 1}. ${t.prompt}`).join('\n'))
|
||||
.replace('{TEAMMATE_HINTS}', teamConfig.suggestedTeammates.map((h, i) => `${i + 1}. ${h}`).join('\n'))
|
||||
.replace('{COMPLETION_PHRASE}', `${phase.id.toUpperCase()}_COMPLETE`);
|
||||
|
||||
// Create a TaskQueue task for the entire phase
|
||||
const queueTask = this.taskQueue.addTask({
|
||||
prompt,
|
||||
workingDir: this.workingDir,
|
||||
priority: 100 - phase.order,
|
||||
completionPhrase: `${phase.id.toUpperCase()}_COMPLETE`,
|
||||
timeoutMs: this.config.phaseTimeoutMs,
|
||||
});
|
||||
|
||||
// Link all phase tasks to this single queue task
|
||||
for (const task of phase.tasks) {
|
||||
task.queueTaskId = queueTask.id;
|
||||
}
|
||||
|
||||
this.persist();
|
||||
this.setupTaskHandlers();
|
||||
|
||||
// Assign the task to the session
|
||||
try {
|
||||
queueTask.assign(session.id);
|
||||
session.assignTask(queueTask.id);
|
||||
this.taskQueue.updateTask(queueTask);
|
||||
await session.sendInput(prompt);
|
||||
} catch (err) {
|
||||
queueTask.fail(getErrorMessage(err));
|
||||
this.taskQueue.updateTask(queueTask);
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
private async assignQueuedTasksToSessions(): Promise<void> {
|
||||
const idleSessions = this.sessionManager.getIdleSessions();
|
||||
const maxSessions =
|
||||
this.getCurrentPhase()?.teamStrategy.type === 'parallel'
|
||||
? (this.getCurrentPhase()?.teamStrategy as { type: 'parallel'; maxSessions: number }).maxSessions
|
||||
: 1;
|
||||
|
||||
const sessionsToUse = idleSessions.slice(0, maxSessions);
|
||||
|
||||
for (const session of sessionsToUse) {
|
||||
const task = this.taskQueue.next();
|
||||
if (!task) break;
|
||||
|
||||
try {
|
||||
task.assign(session.id);
|
||||
session.assignTask(task.id);
|
||||
this.taskQueue.updateTask(task);
|
||||
await session.sendInput(task.prompt);
|
||||
|
||||
this.phaseSessionIds.add(session.id);
|
||||
|
||||
// Find the orchestrator task linked to this queue task
|
||||
const orchTask = this.findOrchestratorTaskByQueueId(task.id);
|
||||
if (orchTask) {
|
||||
orchTask.assignedSessionId = session.id;
|
||||
orchTask.startedAt = Date.now();
|
||||
this.emit('taskAssigned', orchTask, session.id);
|
||||
}
|
||||
} catch (err) {
|
||||
task.fail(getErrorMessage(err));
|
||||
session.clearTask();
|
||||
this.taskQueue.updateTask(task);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Internal — Task Completion Tracking
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
private setupTaskHandlers(): void {
|
||||
this.cleanupTaskHandlers();
|
||||
|
||||
this.sessionCompletionListener = (_sessionId: string, _phrase: string) => {
|
||||
// Session completion — check if it's related to our phase tasks
|
||||
this.checkPhaseCompletion();
|
||||
};
|
||||
|
||||
this.sessionManager.on('sessionCompletion', this.sessionCompletionListener);
|
||||
}
|
||||
|
||||
private cleanupTaskHandlers(): void {
|
||||
if (this.sessionCompletionListener) {
|
||||
this.sessionManager.off('sessionCompletion', this.sessionCompletionListener);
|
||||
this.sessionCompletionListener = null;
|
||||
}
|
||||
}
|
||||
|
||||
private _finalizeTask(queueTaskId: string, status: 'completed' | 'failed', error?: string): OrchestratorTask | null {
|
||||
const orchTask = this.findOrchestratorTaskByQueueId(queueTaskId);
|
||||
if (!orchTask) return null;
|
||||
|
||||
orchTask.status = status;
|
||||
if (status === 'completed') {
|
||||
orchTask.completedAt = Date.now();
|
||||
this.stats.totalTasksCompleted++;
|
||||
} else {
|
||||
orchTask.error = error ?? null;
|
||||
this.stats.totalTasksFailed++;
|
||||
}
|
||||
this.persist();
|
||||
return orchTask;
|
||||
}
|
||||
|
||||
private handleTaskCompleted(queueTaskId: string): void {
|
||||
const orchTask = this._finalizeTask(queueTaskId, 'completed');
|
||||
if (!orchTask) return;
|
||||
|
||||
this.emit('taskCompleted', orchTask);
|
||||
this.checkPhaseCompletion();
|
||||
}
|
||||
|
||||
private handleTaskFailed(queueTaskId: string, error: string): void {
|
||||
const orchTask = this._finalizeTask(queueTaskId, 'failed', error);
|
||||
if (!orchTask) return;
|
||||
|
||||
this.emit('taskFailed', orchTask, error);
|
||||
|
||||
// Check if we should retry the task or fail the phase
|
||||
if (orchTask.retries < 2) {
|
||||
orchTask.retries++;
|
||||
orchTask.status = 'pending';
|
||||
orchTask.error = null;
|
||||
orchTask.queueTaskId = null;
|
||||
// Will be re-queued on next poll
|
||||
} else {
|
||||
this.checkPhaseCompletion();
|
||||
}
|
||||
}
|
||||
|
||||
private startPhasePoll(phase: OrchestratorPhase): void {
|
||||
this.clearPhasePoll();
|
||||
this.phasePollTimer = setInterval(() => {
|
||||
if (this._state !== 'executing') {
|
||||
this.clearPhasePoll();
|
||||
return;
|
||||
}
|
||||
this.pollPhaseStatus(phase);
|
||||
}, PHASE_POLL_INTERVAL_MS);
|
||||
|
||||
// Phase-level timeout — fail the phase if it exceeds the configured timeout
|
||||
this.phaseTimeoutTimer = setTimeout(() => {
|
||||
if (this._state === 'executing' && phase.status === 'executing') {
|
||||
console.warn(`[Orchestrator] Phase "${phase.name}" timed out after ${this.config.phaseTimeoutMs}ms`);
|
||||
this.handlePhaseError(phase, `Phase timed out after ${Math.round(this.config.phaseTimeoutMs / 60000)} minutes`);
|
||||
}
|
||||
}, this.config.phaseTimeoutMs);
|
||||
}
|
||||
|
||||
private _clearTimer(
|
||||
timerKey: 'phasePollTimer' | 'phaseTimeoutTimer' | 'postPhaseTimer',
|
||||
clearFn: typeof clearInterval | typeof clearTimeout
|
||||
): void {
|
||||
if (this[timerKey]) {
|
||||
clearFn(this[timerKey]);
|
||||
this[timerKey] = null;
|
||||
}
|
||||
}
|
||||
|
||||
private clearPhasePoll(): void {
|
||||
this._clearTimer('phasePollTimer', clearInterval);
|
||||
this._clearTimer('phaseTimeoutTimer', clearTimeout);
|
||||
this._clearTimer('postPhaseTimer', clearTimeout);
|
||||
}
|
||||
|
||||
private pollPhaseStatus(phase: OrchestratorPhase): void {
|
||||
// Check for queued tasks that need assignment
|
||||
const pendingTasks = phase.tasks.filter((t) => t.status === 'pending' && !t.queueTaskId);
|
||||
if (pendingTasks.length > 0) {
|
||||
// Re-queue pending tasks
|
||||
for (const task of pendingTasks) {
|
||||
const prompt = this.buildTaskPrompt(task, phase);
|
||||
const queueTask = this.taskQueue.addTask({
|
||||
prompt,
|
||||
workingDir: this.workingDir,
|
||||
priority: 100 - phase.order,
|
||||
completionPhrase: task.completionPhrase,
|
||||
timeoutMs: Math.min(task.timeoutMs, this.config.phaseTimeoutMs),
|
||||
});
|
||||
task.queueTaskId = queueTask.id;
|
||||
task.status = 'running';
|
||||
}
|
||||
this.assignQueuedTasksToSessions().catch(() => {}); // Best effort
|
||||
}
|
||||
|
||||
// Check completion status of queue tasks
|
||||
for (const task of phase.tasks) {
|
||||
if (task.status === 'running' && task.queueTaskId) {
|
||||
const queueTask = this.taskQueue.getTask(task.queueTaskId);
|
||||
if (queueTask) {
|
||||
if (queueTask.isCompleted()) {
|
||||
this.handleTaskCompleted(task.queueTaskId);
|
||||
} else if (queueTask.isFailed()) {
|
||||
this.handleTaskFailed(task.queueTaskId, queueTask.error || 'Task failed');
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
this.checkPhaseCompletion();
|
||||
}
|
||||
|
||||
private checkPhaseCompletion(): void {
|
||||
if (this._state !== 'executing') return;
|
||||
|
||||
const phase = this.getCurrentPhase();
|
||||
if (!phase) return;
|
||||
|
||||
const allDone = phase.tasks.every((t) => t.status === 'completed' || t.status === 'failed');
|
||||
if (!allDone) return;
|
||||
|
||||
const anyFailed = phase.tasks.some((t) => t.status === 'failed');
|
||||
|
||||
this.clearPhasePoll();
|
||||
|
||||
if (anyFailed) {
|
||||
// Phase has failed tasks
|
||||
this.handlePhaseError(phase, 'One or more tasks failed');
|
||||
} else {
|
||||
// All tasks completed — run verification after brief delay
|
||||
this.postPhaseTimer = setTimeout(() => {
|
||||
this.postPhaseTimer = null;
|
||||
this.verifyCurrentPhase().catch((err) => this.handleError(err));
|
||||
}, POST_PHASE_DELAY_MS);
|
||||
}
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Internal — Verification
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
private async verifyCurrentPhase(): Promise<void> {
|
||||
if (!this.plan) return;
|
||||
|
||||
const phase = this.plan.phases[this.currentPhaseIndex];
|
||||
if (!phase) return;
|
||||
|
||||
// Skip verification if no criteria defined
|
||||
if (phase.verificationCriteria.length === 0 && phase.testCommands.length === 0) {
|
||||
phase.status = 'passed';
|
||||
phase.completedAt = Date.now();
|
||||
phase.durationMs = phase.startedAt ? Date.now() - phase.startedAt : null;
|
||||
this.stats.phasesCompleted++;
|
||||
this.persist();
|
||||
this.emit('phaseCompleted', phase);
|
||||
await this.advanceToNextPhase();
|
||||
return;
|
||||
}
|
||||
|
||||
this.setState('verifying');
|
||||
|
||||
// Get a session for verification — wait briefly for sessions to become idle
|
||||
let sessions = this.sessionManager.getIdleSessions();
|
||||
if (sessions.length === 0) {
|
||||
// Wait up to 10s for a session to become idle
|
||||
await new Promise((resolve) => setTimeout(resolve, 10_000));
|
||||
sessions = this.sessionManager.getIdleSessions();
|
||||
}
|
||||
if (sessions.length === 0) {
|
||||
// Still no sessions — log warning and skip verification (don't silently pass)
|
||||
console.warn('[Orchestrator] No idle sessions for verification — skipping (marking passed with warning)');
|
||||
phase.status = 'passed';
|
||||
phase.completedAt = Date.now();
|
||||
phase.durationMs = phase.startedAt ? Date.now() - phase.startedAt : null;
|
||||
this.stats.phasesCompleted++;
|
||||
this.persist();
|
||||
this.emit('phaseCompleted', phase);
|
||||
this.setState('executing');
|
||||
await this.advanceToNextPhase();
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
const result = await this.verifier.verifyPhase(phase, sessions[0]);
|
||||
this.emit('verificationResult', phase, result);
|
||||
|
||||
if (result.passed) {
|
||||
phase.status = 'passed';
|
||||
phase.completedAt = Date.now();
|
||||
phase.durationMs = phase.startedAt ? Date.now() - phase.startedAt : null;
|
||||
this.stats.phasesCompleted++;
|
||||
this.persist();
|
||||
this.emit('phaseCompleted', phase);
|
||||
this.setState('executing');
|
||||
await this.advanceToNextPhase();
|
||||
} else {
|
||||
// Verification failed — attempt replan
|
||||
await this.handleVerificationFailure(phase, result);
|
||||
}
|
||||
} catch (err) {
|
||||
// Verification error — treat as pass (don't block on verification bugs)
|
||||
console.warn('[Orchestrator] Verification error, treating as pass:', err);
|
||||
phase.status = 'passed';
|
||||
phase.completedAt = Date.now();
|
||||
phase.durationMs = phase.startedAt ? Date.now() - phase.startedAt : null;
|
||||
this.stats.phasesCompleted++;
|
||||
this.persist();
|
||||
this.emit('phaseCompleted', phase);
|
||||
this.setState('executing');
|
||||
await this.advanceToNextPhase();
|
||||
}
|
||||
}
|
||||
|
||||
private async handleVerificationFailure(phase: OrchestratorPhase, result: VerificationResult): Promise<void> {
|
||||
if (phase.attempts >= phase.maxAttempts) {
|
||||
// Max retries exceeded
|
||||
phase.status = 'failed';
|
||||
phase.completedAt = Date.now();
|
||||
phase.durationMs = phase.startedAt ? Date.now() - phase.startedAt : null;
|
||||
this.stats.phasesFailed++;
|
||||
this.persist();
|
||||
this.emit('phaseFailed', phase, `Verification failed after ${phase.attempts} attempts: ${result.summary}`);
|
||||
this.setState('failed');
|
||||
return;
|
||||
}
|
||||
|
||||
// Replan and retry
|
||||
this.stats.replanCount++;
|
||||
this.setState('replanning');
|
||||
|
||||
try {
|
||||
await this.replanPhase(phase, result);
|
||||
// Reset task states for retry
|
||||
for (const task of phase.tasks) {
|
||||
task.status = 'pending';
|
||||
task.error = null;
|
||||
task.assignedSessionId = null;
|
||||
task.queueTaskId = null;
|
||||
task.completedAt = null;
|
||||
task.startedAt = null;
|
||||
}
|
||||
phase.status = 'pending';
|
||||
phase.startedAt = null;
|
||||
this.persist();
|
||||
|
||||
this.setState('executing');
|
||||
await this.executeCurrentPhase();
|
||||
} catch (err) {
|
||||
this.handleError(err);
|
||||
}
|
||||
}
|
||||
|
||||
private async replanPhase(phase: OrchestratorPhase, result: VerificationResult): Promise<void> {
|
||||
const completionPhrase = phase.tasks[0]?.completionPhrase || `${phase.id.toUpperCase()}_FIXED`;
|
||||
|
||||
const prompt = REPLAN_PROMPT.replace('{PHASE_NAME}', phase.name)
|
||||
.replace('{ATTEMPT_NUMBER}', String(phase.attempts))
|
||||
.replace('{MAX_ATTEMPTS}', String(phase.maxAttempts))
|
||||
.replace('{FAILURE_SUMMARY}', result.summary)
|
||||
.replace('{SUGGESTIONS}', result.suggestions.join('\n'))
|
||||
.replace('{ORIGINAL_TASKS}', phase.tasks.map((t, i) => `${i + 1}. ${t.prompt}`).join('\n'))
|
||||
.replace('{COMPLETION_PHRASE}', completionPhrase);
|
||||
|
||||
// Create a tracked queue task for the replan (so completion is detected)
|
||||
const queueTask = this.taskQueue.addTask({
|
||||
prompt,
|
||||
workingDir: this.workingDir,
|
||||
priority: 100,
|
||||
completionPhrase,
|
||||
timeoutMs: this.config.phaseTimeoutMs,
|
||||
});
|
||||
|
||||
// Link to first phase task for tracking
|
||||
if (phase.tasks[0]) {
|
||||
phase.tasks[0].queueTaskId = queueTask.id;
|
||||
phase.tasks[0].status = 'running';
|
||||
}
|
||||
|
||||
this.persist();
|
||||
|
||||
// Set up handlers so task completion is tracked
|
||||
this.setupTaskHandlers();
|
||||
|
||||
// Assign to a session
|
||||
const sessions = this.sessionManager.getIdleSessions();
|
||||
if (sessions.length === 0) {
|
||||
console.warn('[Orchestrator] No idle sessions for replan — task queued, will pick up on next poll');
|
||||
// Start polling so the task gets assigned when a session becomes idle
|
||||
this.startPhasePoll(phase);
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
queueTask.assign(sessions[0].id);
|
||||
sessions[0].assignTask(queueTask.id);
|
||||
this.taskQueue.updateTask(queueTask);
|
||||
await sessions[0].sendInput(prompt);
|
||||
} catch (err) {
|
||||
queueTask.fail(getErrorMessage(err));
|
||||
this.taskQueue.updateTask(queueTask);
|
||||
}
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Internal — State Machine
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
/** Read current state (bypasses TypeScript narrowing from guards) */
|
||||
private currentState(): OrchestratorState {
|
||||
return this._state;
|
||||
}
|
||||
|
||||
/** Assert state matches expected or throw */
|
||||
private requireState(...expected: OrchestratorState[]): void {
|
||||
if (!expected.includes(this._state)) {
|
||||
throw new Error(`Expected state "${expected.join('|')}", got "${this._state}"`);
|
||||
}
|
||||
}
|
||||
|
||||
private setState(newState: OrchestratorState): void {
|
||||
const prev = this._state;
|
||||
if (prev === newState) return;
|
||||
this._state = newState;
|
||||
this.persist();
|
||||
this.emit('stateChanged', newState, prev);
|
||||
}
|
||||
|
||||
private async advanceToNextPhase(): Promise<void> {
|
||||
this.currentPhaseIndex++;
|
||||
this.phaseSessionIds.clear();
|
||||
this.persist();
|
||||
|
||||
if (!this.plan || this.currentPhaseIndex >= this.plan.phases.length) {
|
||||
await this.handleCompletion();
|
||||
} else {
|
||||
// Compact between phases if configured
|
||||
if (this.config.compactBetweenPhases) {
|
||||
const sessions = this.sessionManager.getIdleSessions();
|
||||
for (const session of sessions) {
|
||||
try {
|
||||
await session.writeViaMux('/compact');
|
||||
} catch {
|
||||
// Best effort
|
||||
}
|
||||
}
|
||||
// Brief delay for compact to take effect
|
||||
await new Promise((resolve) => setTimeout(resolve, 2000));
|
||||
}
|
||||
|
||||
await this.executeCurrentPhase();
|
||||
}
|
||||
}
|
||||
|
||||
private async handleCompletion(): Promise<void> {
|
||||
this.completedAt = Date.now();
|
||||
this.stats.totalDurationMs = this.startedAt ? this.completedAt - this.startedAt : 0;
|
||||
this.clearPhasePoll();
|
||||
this.cleanupTaskHandlers();
|
||||
this.setState('completed');
|
||||
this.emit('completed', this.stats);
|
||||
}
|
||||
|
||||
private handlePhaseError(phase: OrchestratorPhase, error: string): void {
|
||||
if (phase.attempts >= phase.maxAttempts) {
|
||||
phase.status = 'failed';
|
||||
phase.completedAt = Date.now();
|
||||
phase.durationMs = phase.startedAt ? Date.now() - phase.startedAt : null;
|
||||
this.stats.phasesFailed++;
|
||||
this.persist();
|
||||
this.emit('phaseFailed', phase, error);
|
||||
this.setState('failed');
|
||||
} else {
|
||||
// Retry the phase
|
||||
for (const task of phase.tasks) {
|
||||
if (task.status === 'failed') {
|
||||
task.status = 'pending';
|
||||
task.error = null;
|
||||
task.queueTaskId = null;
|
||||
task.assignedSessionId = null;
|
||||
}
|
||||
}
|
||||
phase.status = 'pending';
|
||||
this.persist();
|
||||
this.executeCurrentPhase().catch((err) => this.handleError(err));
|
||||
}
|
||||
}
|
||||
|
||||
private handleError(err: unknown): void {
|
||||
const error = err instanceof Error ? err : new Error(getErrorMessage(err));
|
||||
console.error('[Orchestrator] Error:', error.message);
|
||||
this.setState('failed');
|
||||
this.emit('error', error);
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Internal — Persistence
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
private persist(): void {
|
||||
this.store.setOrchestratorState(this.getStatus());
|
||||
}
|
||||
|
||||
private restore(): void {
|
||||
const saved = this.store.getOrchestratorState();
|
||||
if (!saved) return;
|
||||
|
||||
// If we crashed while running, reset to failed
|
||||
if (saved.state === 'executing' || saved.state === 'verifying' || saved.state === 'replanning') {
|
||||
this._state = 'failed';
|
||||
this.plan = saved.plan;
|
||||
this.currentPhaseIndex = saved.currentPhaseIndex;
|
||||
this.startedAt = saved.startedAt;
|
||||
this.config = saved.config;
|
||||
this.stats = saved.stats;
|
||||
this.store.setOrchestratorState({ ...saved, state: 'failed' });
|
||||
} else if (saved.state === 'planning' || saved.state === 'approval') {
|
||||
// Planning/approval — reset to idle (plan is lost)
|
||||
this.store.clearOrchestratorState();
|
||||
} else if (saved.state === 'completed' || saved.state === 'failed') {
|
||||
// Preserve completed/failed state for UI display
|
||||
this._state = saved.state;
|
||||
this.plan = saved.plan;
|
||||
this.currentPhaseIndex = saved.currentPhaseIndex;
|
||||
this.startedAt = saved.startedAt;
|
||||
this.completedAt = saved.completedAt;
|
||||
this.config = saved.config;
|
||||
this.stats = saved.stats;
|
||||
}
|
||||
}
|
||||
|
||||
private reset(): void {
|
||||
this._state = 'idle';
|
||||
this.plan = null;
|
||||
this.currentPhaseIndex = 0;
|
||||
this.startedAt = null;
|
||||
this.completedAt = null;
|
||||
this.stats = createInitialOrchestratorStats();
|
||||
this.pausedState = null;
|
||||
this.phaseSessionIds.clear();
|
||||
this.clearPhasePoll();
|
||||
this.cleanupTaskHandlers();
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Internal — Helpers
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
private buildTaskPrompt(task: OrchestratorTask, phase: OrchestratorPhase): string {
|
||||
if (phase.tasks.length === 1) {
|
||||
// Single task — use simpler prompt
|
||||
const completedPhases = this.getCompletedPhasesSummary();
|
||||
return SINGLE_TASK_PROMPT.replace('{TASK}', task.prompt)
|
||||
.replace('{GOAL}', this.plan?.goal || '')
|
||||
.replace('{CONTEXT}', completedPhases ? `Previous phases completed: ${completedPhases}` : '')
|
||||
.replace('{COMPLETION_PHRASE}', task.completionPhrase);
|
||||
}
|
||||
|
||||
// Multi-task phase — use full prompt
|
||||
return PHASE_EXECUTION_PROMPT.replace('{PHASE_NAME}', phase.name)
|
||||
.replace('{GOAL}', this.plan?.goal || '')
|
||||
.replace('{COMPLETED_PHASES}', this.getCompletedPhasesSummary() || 'None yet')
|
||||
.replace('{TASK_LIST}', phase.tasks.map((t, i) => `${i + 1}. ${t.prompt}`).join('\n'))
|
||||
.replace('{VERIFICATION_CRITERIA}', phase.verificationCriteria.join('\n') || 'No specific criteria')
|
||||
.replace('{COMPLETION_PHRASE}', task.completionPhrase);
|
||||
}
|
||||
|
||||
private getCompletedPhasesSummary(): string {
|
||||
if (!this.plan) return '';
|
||||
return this.plan.phases
|
||||
.filter((p) => p.status === 'passed' || p.status === 'skipped')
|
||||
.map((p) => `${p.name}: ${p.status}`)
|
||||
.join(', ');
|
||||
}
|
||||
|
||||
private findOrchestratorTaskByQueueId(queueTaskId: string): OrchestratorTask | null {
|
||||
if (!this.plan) return null;
|
||||
for (const phase of this.plan.phases) {
|
||||
for (const task of phase.tasks) {
|
||||
if (task.queueTaskId === queueTaskId) return task;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Clean up resources when the loop is being destroyed. */
|
||||
destroy(): void {
|
||||
this.clearPhasePoll();
|
||||
this.cleanupTaskHandlers();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,412 @@
|
||||
/**
|
||||
* @fileoverview Orchestrator plan generation — converts goals into phased plans.
|
||||
*
|
||||
* Wraps PlanOrchestrator for AI-powered plan generation, then groups the
|
||||
* resulting PlanItems into sequential phases with team strategies and
|
||||
* verification criteria.
|
||||
*
|
||||
* Phase grouping algorithm:
|
||||
* 1. Topological sort by dependencies (Kahn's algorithm)
|
||||
* 2. Group into dependency layers
|
||||
* 3. Sub-group by TDD phase within layers
|
||||
* 4. Merge small adjacent phases
|
||||
* 5. Assign team strategies based on parallelism potential
|
||||
*
|
||||
* Key exports:
|
||||
* - `OrchestratorPlanner` class — plan generation + phase grouping
|
||||
*
|
||||
* @dependencies plan-orchestrator (AI plan generation), types (OrchestratorPlan, PlanItem)
|
||||
* @consumedby orchestrator-loop
|
||||
*
|
||||
* @module orchestrator-planner
|
||||
*/
|
||||
|
||||
import { v4 as uuidv4 } from 'uuid';
|
||||
import { PlanOrchestrator, type DetailedPlanResult, type ProgressCallback } from './plan-orchestrator.js';
|
||||
import type { TerminalMultiplexer } from './mux-interface.js';
|
||||
import type {
|
||||
PlanItem,
|
||||
TddPhase,
|
||||
OrchestratorPlan,
|
||||
OrchestratorPhase,
|
||||
OrchestratorTask,
|
||||
OrchestratorConfig,
|
||||
TeamStrategy,
|
||||
PhaseStatus,
|
||||
} from './types.js';
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Constants
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
/** Maximum number of phases (prevents runaway plans) */
|
||||
const MAX_PHASES = 10;
|
||||
|
||||
/** Maximum total tasks across all phases */
|
||||
const MAX_TOTAL_TASKS = 50;
|
||||
|
||||
/** Default task timeout (10 minutes) */
|
||||
const DEFAULT_TASK_TIMEOUT_MS = 10 * 60 * 1000;
|
||||
|
||||
/** Minimum tasks in a phase before it gets merged with adjacent */
|
||||
const MIN_PHASE_TASKS = 2;
|
||||
|
||||
/** TDD phase ordering for grouping */
|
||||
const TDD_PHASE_ORDER: Record<TddPhase, number> = {
|
||||
setup: 0,
|
||||
test: 1,
|
||||
impl: 2,
|
||||
verify: 3,
|
||||
review: 4,
|
||||
};
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// OrchestratorPlanner
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
export class OrchestratorPlanner {
|
||||
private mux: TerminalMultiplexer;
|
||||
private workingDir: string;
|
||||
private config: OrchestratorConfig;
|
||||
private orchestrator: PlanOrchestrator | null = null;
|
||||
|
||||
constructor(mux: TerminalMultiplexer, workingDir: string, config: OrchestratorConfig) {
|
||||
this.mux = mux;
|
||||
this.workingDir = workingDir;
|
||||
this.config = config;
|
||||
}
|
||||
|
||||
/**
|
||||
* Generate a phased plan from a user goal.
|
||||
*
|
||||
* Uses PlanOrchestrator for AI plan generation, then groups results into phases.
|
||||
*/
|
||||
async generatePlan(goal: string, onProgress?: ProgressCallback): Promise<OrchestratorPlan> {
|
||||
const startTime = Date.now();
|
||||
|
||||
// Create a PlanOrchestrator for this plan generation
|
||||
this.orchestrator = new PlanOrchestrator(this.mux, this.workingDir, undefined, {
|
||||
defaultModel: this.config.plannerModel,
|
||||
});
|
||||
|
||||
try {
|
||||
onProgress?.('planning', 'Generating detailed plan...');
|
||||
|
||||
const result: DetailedPlanResult = await this.orchestrator.generateDetailedPlan(goal, onProgress);
|
||||
|
||||
if (!result.success || !result.items || result.items.length === 0) {
|
||||
throw new Error(result.error || 'Plan generation returned no items');
|
||||
}
|
||||
|
||||
// Cap total tasks
|
||||
const items = result.items.slice(0, MAX_TOTAL_TASKS);
|
||||
|
||||
onProgress?.('grouping', 'Organizing plan into phases...');
|
||||
|
||||
// Group items into phases
|
||||
const phases = this.groupIntoPhases(items, goal);
|
||||
|
||||
// Assign team strategies
|
||||
this.assignTeamStrategies(phases);
|
||||
|
||||
// Generate unique completion phrases
|
||||
this.generateCompletionPhrases(phases);
|
||||
|
||||
const plan: OrchestratorPlan = {
|
||||
id: uuidv4(),
|
||||
goal,
|
||||
createdAt: Date.now(),
|
||||
phases,
|
||||
metadata: {
|
||||
totalTasks: phases.reduce((sum, p) => sum + p.tasks.length, 0),
|
||||
estimatedComplexity: this.estimateComplexity(items),
|
||||
modelUsed: this.config.plannerModel,
|
||||
planDurationMs: Date.now() - startTime,
|
||||
},
|
||||
};
|
||||
|
||||
return plan;
|
||||
} finally {
|
||||
this.orchestrator = null;
|
||||
}
|
||||
}
|
||||
|
||||
/** Cancel in-progress plan generation. */
|
||||
async cancel(): Promise<void> {
|
||||
if (this.orchestrator) {
|
||||
await this.orchestrator.cancel();
|
||||
this.orchestrator = null;
|
||||
}
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Phase Grouping
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
/**
|
||||
* Group PlanItems into sequential phases.
|
||||
*
|
||||
* Algorithm:
|
||||
* 1. Build dependency graph and assign IDs to items without them
|
||||
* 2. Topological sort into dependency layers (Kahn's algorithm)
|
||||
* 3. Sub-group within each layer by TDD phase
|
||||
* 4. Merge small phases with their neighbors
|
||||
*/
|
||||
private groupIntoPhases(items: PlanItem[], _goal: string): OrchestratorPhase[] {
|
||||
// Ensure all items have IDs
|
||||
const indexedItems = items.map((item, i) => ({
|
||||
...item,
|
||||
id: item.id || `task-${i}`,
|
||||
}));
|
||||
|
||||
// Build adjacency and in-degree for Kahn's algorithm
|
||||
const idSet = new Set(indexedItems.map((item) => item.id!));
|
||||
const inDegree = new Map<string, number>();
|
||||
const dependents = new Map<string, string[]>(); // id → items that depend on it
|
||||
|
||||
for (const item of indexedItems) {
|
||||
inDegree.set(item.id!, 0);
|
||||
dependents.set(item.id!, []);
|
||||
}
|
||||
|
||||
for (const item of indexedItems) {
|
||||
const deps = (item.dependencies || []).filter((d) => idSet.has(d));
|
||||
inDegree.set(item.id!, deps.length);
|
||||
for (const dep of deps) {
|
||||
dependents.get(dep)!.push(item.id!);
|
||||
}
|
||||
}
|
||||
|
||||
// Kahn's algorithm — produce dependency layers
|
||||
const layers: PlanItem[][] = [];
|
||||
const remaining = new Set(indexedItems.map((item) => item.id!));
|
||||
|
||||
while (remaining.size > 0) {
|
||||
// Find items with no remaining dependencies (in-degree 0)
|
||||
const layer: PlanItem[] = [];
|
||||
for (const id of remaining) {
|
||||
if (inDegree.get(id)! === 0) {
|
||||
layer.push(indexedItems.find((item) => item.id === id)!);
|
||||
}
|
||||
}
|
||||
|
||||
if (layer.length === 0) {
|
||||
// Circular dependency — add all remaining items as a single layer
|
||||
for (const id of remaining) {
|
||||
layer.push(indexedItems.find((item) => item.id === id)!);
|
||||
}
|
||||
}
|
||||
|
||||
layers.push(layer);
|
||||
|
||||
// Remove this layer's items and update in-degrees
|
||||
for (const item of layer) {
|
||||
remaining.delete(item.id!);
|
||||
for (const dep of dependents.get(item.id!) || []) {
|
||||
if (remaining.has(dep)) {
|
||||
inDegree.set(dep, Math.max(0, inDegree.get(dep)! - 1));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Sub-group each layer by TDD phase
|
||||
const rawPhases: PlanItem[][] = [];
|
||||
for (const layer of layers) {
|
||||
const byPhase = new Map<string, PlanItem[]>();
|
||||
for (const item of layer) {
|
||||
const phase = item.tddPhase || 'impl';
|
||||
if (!byPhase.has(phase)) byPhase.set(phase, []);
|
||||
byPhase.get(phase)!.push(item);
|
||||
}
|
||||
|
||||
// Sort sub-groups by TDD phase order
|
||||
const sorted = [...byPhase.entries()].sort(
|
||||
([a], [b]) => (TDD_PHASE_ORDER[a as TddPhase] ?? 2) - (TDD_PHASE_ORDER[b as TddPhase] ?? 2)
|
||||
);
|
||||
|
||||
for (const [, items] of sorted) {
|
||||
rawPhases.push(items);
|
||||
}
|
||||
}
|
||||
|
||||
// Merge small phases with their previous neighbor
|
||||
const mergedPhases: PlanItem[][] = [];
|
||||
for (const phase of rawPhases) {
|
||||
if (mergedPhases.length > 0 && phase.length < MIN_PHASE_TASKS) {
|
||||
const prev = mergedPhases[mergedPhases.length - 1];
|
||||
if (prev.length < MIN_PHASE_TASKS) {
|
||||
// Merge with previous
|
||||
prev.push(...phase);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
mergedPhases.push([...phase]);
|
||||
}
|
||||
|
||||
// Cap at MAX_PHASES by merging tail phases
|
||||
while (mergedPhases.length > MAX_PHASES) {
|
||||
const last = mergedPhases.pop()!;
|
||||
mergedPhases[mergedPhases.length - 1].push(...last);
|
||||
}
|
||||
|
||||
// Convert to OrchestratorPhase objects
|
||||
return mergedPhases.map((phaseItems, index) => this.createPhase(phaseItems, index));
|
||||
}
|
||||
|
||||
private createPhase(items: PlanItem[], order: number): OrchestratorPhase {
|
||||
// Derive phase name from TDD phases and priorities
|
||||
const tddPhases = [...new Set(items.map((i) => i.tddPhase).filter(Boolean))];
|
||||
const name = this.generatePhaseName(items, tddPhases as TddPhase[], order);
|
||||
const description = items.map((i) => i.content).join('; ');
|
||||
|
||||
const tasks: OrchestratorTask[] = items.map((item, i) => ({
|
||||
id: `phase-${order + 1}-task-${i + 1}`,
|
||||
phaseId: `phase-${order + 1}`,
|
||||
prompt: item.content,
|
||||
status: 'pending' as const,
|
||||
assignedSessionId: null,
|
||||
queueTaskId: null,
|
||||
parallel: items.length > 1, // Tasks within a phase are parallel by default
|
||||
completionPhrase: '', // Assigned later
|
||||
timeoutMs: DEFAULT_TASK_TIMEOUT_MS,
|
||||
startedAt: null,
|
||||
completedAt: null,
|
||||
error: null,
|
||||
retries: 0,
|
||||
}));
|
||||
|
||||
// Extract verification criteria and test commands from items
|
||||
const verificationCriteria = items
|
||||
.map((i) => i.verificationCriteria)
|
||||
.filter((v): v is string => v != null && v.length > 0);
|
||||
|
||||
const testCommands = items.map((i) => i.testCommand).filter((t): t is string => t != null && t.length > 0);
|
||||
|
||||
return {
|
||||
id: `phase-${order + 1}`,
|
||||
name,
|
||||
description,
|
||||
order,
|
||||
status: 'pending' as PhaseStatus,
|
||||
tasks,
|
||||
verificationCriteria,
|
||||
testCommands,
|
||||
maxAttempts: this.config.maxPhaseRetries,
|
||||
attempts: 0,
|
||||
startedAt: null,
|
||||
completedAt: null,
|
||||
durationMs: null,
|
||||
teamStrategy: { type: 'single' }, // Assigned later
|
||||
};
|
||||
}
|
||||
|
||||
private generatePhaseName(items: PlanItem[], tddPhases: TddPhase[], order: number): string {
|
||||
// Try to create a meaningful name based on content
|
||||
const priorities = [...new Set(items.map((i) => i.priority).filter(Boolean))];
|
||||
|
||||
if (tddPhases.length === 1) {
|
||||
const phaseNames: Record<TddPhase, string> = {
|
||||
setup: 'Setup & Configuration',
|
||||
test: 'Test Definition',
|
||||
impl: 'Implementation',
|
||||
verify: 'Verification',
|
||||
review: 'Review & Polish',
|
||||
};
|
||||
return `Phase ${order + 1}: ${phaseNames[tddPhases[0]]}`;
|
||||
}
|
||||
|
||||
if (priorities.includes('P0') && priorities.length === 1) {
|
||||
return `Phase ${order + 1}: Critical Foundation`;
|
||||
}
|
||||
|
||||
return `Phase ${order + 1}: ${items.length > 1 ? 'Parallel Tasks' : items[0].content.slice(0, 50)}`;
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Team Strategy Assignment
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
private assignTeamStrategies(phases: OrchestratorPhase[]): void {
|
||||
for (const phase of phases) {
|
||||
phase.teamStrategy = this.computeTeamStrategy(phase);
|
||||
}
|
||||
}
|
||||
|
||||
private computeTeamStrategy(phase: OrchestratorPhase): TeamStrategy {
|
||||
const taskCount = phase.tasks.length;
|
||||
const parallelTasks = phase.tasks.filter((t) => t.parallel).length;
|
||||
|
||||
// Single task or no parallel potential → single session
|
||||
if (taskCount <= 2 || parallelTasks <= 1) {
|
||||
return { type: 'single' };
|
||||
}
|
||||
|
||||
// If team agents are disabled, use parallel sessions instead
|
||||
if (!this.config.enableTeamAgents) {
|
||||
return {
|
||||
type: 'parallel',
|
||||
maxSessions: Math.min(parallelTasks, this.config.maxParallelSessions),
|
||||
};
|
||||
}
|
||||
|
||||
// 4+ parallel tasks with team agents enabled → team mode
|
||||
if (parallelTasks >= 4) {
|
||||
return {
|
||||
type: 'team',
|
||||
config: {
|
||||
leadPrompt: this.buildTeamLeadPrompt(phase),
|
||||
suggestedTeammates: phase.tasks.slice(0, 4).map((t) => `Specialist for: ${t.prompt.slice(0, 80)}`),
|
||||
maxTeammates: Math.min(parallelTasks, 4),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// 3 parallel tasks → parallel sessions
|
||||
return {
|
||||
type: 'parallel',
|
||||
maxSessions: Math.min(parallelTasks, this.config.maxParallelSessions),
|
||||
};
|
||||
}
|
||||
|
||||
private buildTeamLeadPrompt(phase: OrchestratorPhase): string {
|
||||
const taskList = phase.tasks.map((t, i) => `${i + 1}. ${t.prompt}`).join('\n');
|
||||
|
||||
return [
|
||||
`You are the team lead for "${phase.name}".`,
|
||||
`Create teammates and delegate the following tasks for parallel execution:`,
|
||||
'',
|
||||
taskList,
|
||||
'',
|
||||
`Each teammate should focus on one task area.`,
|
||||
`When all tasks are complete, verify the results and output: <promise>${phase.id.toUpperCase()}_COMPLETE</promise>`,
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Completion Phrases
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
private generateCompletionPhrases(phases: OrchestratorPhase[]): void {
|
||||
for (const phase of phases) {
|
||||
for (const task of phase.tasks) {
|
||||
// Generate a unique, deterministic completion phrase per task
|
||||
task.completionPhrase = `ORCH_P${phase.order + 1}_T${phase.tasks.indexOf(task) + 1}`;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Helpers
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
private estimateComplexity(items: PlanItem[]): 'low' | 'medium' | 'high' {
|
||||
const total = items.length;
|
||||
const highComplexity = items.filter((i) => i.complexity === 'high').length;
|
||||
const p0Count = items.filter((i) => i.priority === 'P0').length;
|
||||
|
||||
if (total > 20 || highComplexity > 5 || p0Count > 8) return 'high';
|
||||
if (total > 10 || highComplexity > 2 || p0Count > 4) return 'medium';
|
||||
return 'low';
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,298 @@
|
||||
/**
|
||||
* @fileoverview Orchestrator phase verification.
|
||||
*
|
||||
* Runs verification checks after each phase completes:
|
||||
* - Test commands (shell commands via session)
|
||||
* - AI review (ask Claude to evaluate phase results)
|
||||
*
|
||||
* Three verification modes:
|
||||
* - strict: ALL test commands must pass AND AI review must approve
|
||||
* - moderate: Test commands must pass, AI review is advisory
|
||||
* - lenient: At least one test command passes, AI review skipped
|
||||
*
|
||||
* Key exports:
|
||||
* - `OrchestratorVerifier` class — phase verification engine
|
||||
*
|
||||
* @dependencies types (OrchestratorPhase, VerificationResult, VerificationCheck, OrchestratorConfig)
|
||||
* @consumedby orchestrator-loop
|
||||
*
|
||||
* @module orchestrator-verifier
|
||||
*/
|
||||
|
||||
import type { Session } from './session.js';
|
||||
import {
|
||||
getErrorMessage,
|
||||
type OrchestratorPhase,
|
||||
type OrchestratorConfig,
|
||||
type VerificationResult,
|
||||
type VerificationCheck,
|
||||
} from './types.js';
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Constants
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
/** Timeout for individual test command execution (2 minutes) */
|
||||
const TEST_COMMAND_TIMEOUT_MS = 2 * 60 * 1000;
|
||||
|
||||
/** Timeout for AI review (3 minutes) */
|
||||
const AI_REVIEW_TIMEOUT_MS = 3 * 60 * 1000;
|
||||
|
||||
/** Completion phrase for AI verification pass */
|
||||
const VERIFY_PASS_PHRASE = 'ORCH_VERIFY_PASS';
|
||||
|
||||
/** Completion phrase for AI verification fail */
|
||||
const VERIFY_FAIL_PHRASE = 'ORCH_VERIFY_FAIL';
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// OrchestratorVerifier
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
export class OrchestratorVerifier {
|
||||
private config: OrchestratorConfig;
|
||||
|
||||
constructor(config: OrchestratorConfig) {
|
||||
this.config = config;
|
||||
}
|
||||
|
||||
/**
|
||||
* Run all verification checks for a completed phase.
|
||||
*
|
||||
* @param phase - The phase to verify
|
||||
* @param session - Session to use for running commands/reviews
|
||||
* @returns Verification result with pass/fail and suggestions
|
||||
*/
|
||||
async verifyPhase(phase: OrchestratorPhase, session: Session): Promise<VerificationResult> {
|
||||
const checks: VerificationCheck[] = [];
|
||||
const mode = this.config.verificationMode;
|
||||
|
||||
// Skip verification entirely in lenient mode with no test commands
|
||||
if (mode === 'lenient' && phase.testCommands.length === 0 && phase.verificationCriteria.length === 0) {
|
||||
return {
|
||||
passed: true,
|
||||
checks: [],
|
||||
summary: 'Verification skipped (lenient mode, no checks defined)',
|
||||
suggestions: [],
|
||||
};
|
||||
}
|
||||
|
||||
// Run test commands if any are defined
|
||||
if (phase.testCommands.length > 0) {
|
||||
const testChecks = await this.runTestCommands(phase.testCommands, session);
|
||||
checks.push(...testChecks);
|
||||
}
|
||||
|
||||
// Run AI review in strict and moderate modes
|
||||
if (mode !== 'lenient' && phase.verificationCriteria.length > 0) {
|
||||
const aiCheck = await this.aiReview(phase, session);
|
||||
checks.push(aiCheck);
|
||||
}
|
||||
|
||||
// Determine pass/fail based on mode
|
||||
const passed = this.evaluateChecks(checks, mode);
|
||||
|
||||
// Generate suggestions for failed checks
|
||||
const suggestions = this.generateSuggestions(checks, phase);
|
||||
|
||||
const passedCount = checks.filter((c) => c.passed).length;
|
||||
const summary =
|
||||
checks.length === 0 ? 'No verification checks defined' : `${passedCount}/${checks.length} checks passed`;
|
||||
|
||||
return { passed, checks, summary, suggestions };
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Test Command Execution
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
private async runTestCommands(commands: string[], session: Session): Promise<VerificationCheck[]> {
|
||||
const checks: VerificationCheck[] = [];
|
||||
|
||||
for (const command of commands) {
|
||||
try {
|
||||
const check = await this.runSingleTestCommand(command, session);
|
||||
checks.push(check);
|
||||
} catch (err) {
|
||||
checks.push({
|
||||
type: 'test_command',
|
||||
description: `Run: ${command}`,
|
||||
passed: false,
|
||||
output: getErrorMessage(err),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
return checks;
|
||||
}
|
||||
|
||||
private async runSingleTestCommand(command: string, session: Session): Promise<VerificationCheck> {
|
||||
// Send the test command to the session and wait for completion
|
||||
// We use a unique marker to detect when the command finishes
|
||||
const marker = `ORCH_TEST_${Date.now()}`;
|
||||
const wrappedCommand = `${command} && echo ${marker}_PASS || echo ${marker}_FAIL`;
|
||||
|
||||
const result = await this.sendAndWaitForMarker(session, wrappedCommand, marker, TEST_COMMAND_TIMEOUT_MS);
|
||||
|
||||
return {
|
||||
type: 'test_command',
|
||||
description: `Run: ${command}`,
|
||||
passed: result.includes(`${marker}_PASS`),
|
||||
output: result.slice(0, 2000), // Truncate output
|
||||
};
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// AI Review
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
private async aiReview(phase: OrchestratorPhase, session: Session): Promise<VerificationCheck> {
|
||||
const prompt = this.buildVerificationPrompt(phase);
|
||||
|
||||
try {
|
||||
const result = await this.sendAndWaitForMarker(
|
||||
session,
|
||||
prompt,
|
||||
VERIFY_PASS_PHRASE,
|
||||
AI_REVIEW_TIMEOUT_MS,
|
||||
VERIFY_FAIL_PHRASE
|
||||
);
|
||||
|
||||
const passed = result.includes(VERIFY_PASS_PHRASE);
|
||||
|
||||
return {
|
||||
type: 'ai_review',
|
||||
description: `AI review of "${phase.name}"`,
|
||||
passed,
|
||||
output: result.slice(0, 3000),
|
||||
};
|
||||
} catch (err) {
|
||||
return {
|
||||
type: 'ai_review',
|
||||
description: `AI review of "${phase.name}"`,
|
||||
passed: false,
|
||||
output: `AI review timed out or failed: ${getErrorMessage(err)}`,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private buildVerificationPrompt(phase: OrchestratorPhase): string {
|
||||
const criteria = phase.verificationCriteria.map((c, i) => `${i + 1}. ${c}`).join('\n');
|
||||
|
||||
return [
|
||||
`Review the work done in "${phase.name}". Check these criteria:`,
|
||||
'',
|
||||
criteria,
|
||||
'',
|
||||
`If ALL criteria are met, respond with: ${VERIFY_PASS_PHRASE}`,
|
||||
`If ANY criteria fail, respond with: ${VERIFY_FAIL_PHRASE} and explain what failed.`,
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Evaluation
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
private evaluateChecks(checks: VerificationCheck[], mode: OrchestratorConfig['verificationMode']): boolean {
|
||||
if (checks.length === 0) return true;
|
||||
|
||||
const testChecks = checks.filter((c) => c.type === 'test_command');
|
||||
const aiChecks = checks.filter((c) => c.type === 'ai_review');
|
||||
|
||||
switch (mode) {
|
||||
case 'strict':
|
||||
// ALL checks must pass
|
||||
return checks.every((c) => c.passed);
|
||||
|
||||
case 'moderate':
|
||||
// All test commands must pass; AI review is advisory
|
||||
return testChecks.length === 0 || testChecks.every((c) => c.passed);
|
||||
|
||||
case 'lenient':
|
||||
// At least one test passes (AI review skipped in lenient mode)
|
||||
return testChecks.length === 0 || testChecks.some((c) => c.passed);
|
||||
|
||||
default:
|
||||
return aiChecks.every((c) => c.passed) && testChecks.every((c) => c.passed);
|
||||
}
|
||||
}
|
||||
|
||||
private generateSuggestions(checks: VerificationCheck[], phase: OrchestratorPhase): string[] {
|
||||
const suggestions: string[] = [];
|
||||
const failedChecks = checks.filter((c) => !c.passed);
|
||||
|
||||
if (failedChecks.length === 0) return suggestions;
|
||||
|
||||
for (const check of failedChecks) {
|
||||
if (check.type === 'test_command') {
|
||||
suggestions.push(`Fix failing test: ${check.description}`);
|
||||
} else if (check.type === 'ai_review' && check.output) {
|
||||
// Extract failure reasons from AI review output
|
||||
suggestions.push(`Address AI review feedback for "${phase.name}"`);
|
||||
}
|
||||
}
|
||||
|
||||
return suggestions;
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Session Communication
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
/**
|
||||
* Send a prompt to a session and wait for a marker phrase in the output.
|
||||
*
|
||||
* @param session - Session to send to
|
||||
* @param input - Prompt/command to send
|
||||
* @param marker - Primary marker to watch for
|
||||
* @param timeoutMs - Maximum wait time
|
||||
* @param altMarker - Alternative marker (for pass/fail detection)
|
||||
* @returns Captured output containing the marker
|
||||
*/
|
||||
private sendAndWaitForMarker(
|
||||
session: Session,
|
||||
input: string,
|
||||
marker: string,
|
||||
timeoutMs: number,
|
||||
altMarker?: string
|
||||
): Promise<string> {
|
||||
return new Promise<string>((resolve, reject) => {
|
||||
let output = '';
|
||||
let resolved = false;
|
||||
|
||||
const timer = setTimeout(() => {
|
||||
if (!resolved) {
|
||||
resolved = true;
|
||||
cleanup();
|
||||
reject(new Error(`Timeout waiting for marker "${marker}" after ${timeoutMs}ms`));
|
||||
}
|
||||
}, timeoutMs);
|
||||
|
||||
const handler = (data: string) => {
|
||||
if (resolved) return;
|
||||
output += data;
|
||||
|
||||
if (output.includes(marker) || (altMarker && output.includes(altMarker))) {
|
||||
resolved = true;
|
||||
cleanup();
|
||||
resolve(output);
|
||||
}
|
||||
};
|
||||
|
||||
const cleanup = () => {
|
||||
clearTimeout(timer);
|
||||
session.off('terminal', handler);
|
||||
};
|
||||
|
||||
session.on('terminal', handler);
|
||||
|
||||
// Send the input
|
||||
session.sendInput(input).catch((err) => {
|
||||
if (!resolved) {
|
||||
resolved = true;
|
||||
cleanup();
|
||||
reject(err);
|
||||
}
|
||||
});
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -20,7 +20,7 @@ import type { TerminalMultiplexer } from './mux-interface.js';
|
||||
import { existsSync, mkdirSync, writeFileSync } from 'node:fs';
|
||||
import { join } from 'node:path';
|
||||
import { RESEARCH_AGENT_PROMPT, PLANNER_PROMPT } from './prompts/index.js';
|
||||
import type { PlanItem } from './types.js';
|
||||
import { getErrorMessage, type PlanItem } from './types.js';
|
||||
|
||||
// Re-export for backward compatibility
|
||||
export type { PlanItem };
|
||||
@@ -68,7 +68,7 @@ export interface DetailedPlanResult {
|
||||
|
||||
export type ProgressCallback = (phase: string, detail: string) => void;
|
||||
|
||||
export interface PlanSubagentEvent {
|
||||
interface PlanSubagentEvent {
|
||||
type: 'started' | 'progress' | 'completed' | 'failed';
|
||||
agentId: string;
|
||||
agentType: 'research' | 'planner';
|
||||
@@ -80,7 +80,7 @@ export interface PlanSubagentEvent {
|
||||
error?: string;
|
||||
}
|
||||
|
||||
export type SubagentCallback = (event: PlanSubagentEvent) => void;
|
||||
type SubagentCallback = (event: PlanSubagentEvent) => void;
|
||||
|
||||
// ============================================================================
|
||||
// JSON Repair Helper
|
||||
@@ -231,6 +231,49 @@ export class PlanOrchestrator {
|
||||
return md;
|
||||
}
|
||||
|
||||
private _extractJsonFromResponse(response: string): string | null {
|
||||
let jsonMatch = response.match(/```(?:json)?\s*(\{[\s\S]*?\})\s*```/);
|
||||
if (jsonMatch) {
|
||||
jsonMatch = [jsonMatch[1]]; // Use captured group (inside code block)
|
||||
} else {
|
||||
jsonMatch = response.match(/\{[\s\S]*\}/);
|
||||
}
|
||||
return jsonMatch ? jsonMatch[0] : null;
|
||||
}
|
||||
|
||||
private _emitAgentFailure(
|
||||
onSubagent: SubagentCallback | undefined,
|
||||
agentId: string,
|
||||
agentType: 'research' | 'planner',
|
||||
model: string,
|
||||
error: string,
|
||||
durationMs: number
|
||||
): void {
|
||||
onSubagent?.({
|
||||
type: 'failed',
|
||||
agentId,
|
||||
agentType,
|
||||
model,
|
||||
status: 'failed',
|
||||
error,
|
||||
durationMs,
|
||||
});
|
||||
}
|
||||
|
||||
private _formatResearchSection(
|
||||
parts: string[],
|
||||
title: string,
|
||||
items: unknown[],
|
||||
formatter: (item: unknown) => string[]
|
||||
): void {
|
||||
if (items.length === 0) return;
|
||||
parts.push(title);
|
||||
for (const item of items.slice(0, 5)) {
|
||||
parts.push(...formatter(item));
|
||||
}
|
||||
parts.push('');
|
||||
}
|
||||
|
||||
async cancel(): Promise<void> {
|
||||
this.cancelled = true;
|
||||
// Stop all running sessions and await cleanup to prevent PTY process leaks
|
||||
@@ -312,7 +355,7 @@ export class PlanOrchestrator {
|
||||
} catch (err) {
|
||||
return {
|
||||
success: false,
|
||||
error: err instanceof Error ? err.message : String(err),
|
||||
error: getErrorMessage(err),
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -322,32 +365,23 @@ export class PlanOrchestrator {
|
||||
|
||||
const parts: string[] = ['## Research Context\n'];
|
||||
|
||||
if (research.findings.externalResources.length > 0) {
|
||||
parts.push('### External Resources');
|
||||
for (const r of research.findings.externalResources.slice(0, 5)) {
|
||||
parts.push(`- ${r.title}${r.url ? ` (${r.url})` : ''}`);
|
||||
if (r.keyInsights.length > 0) {
|
||||
parts.push(` Key insights: ${r.keyInsights.slice(0, 3).join(', ')}`);
|
||||
}
|
||||
this._formatResearchSection(parts, '### External Resources', research.findings.externalResources, (item) => {
|
||||
const r = item as ResearchResult['findings']['externalResources'][number];
|
||||
const lines = [`- ${r.title}${r.url ? ` (${r.url})` : ''}`];
|
||||
if (r.keyInsights.length > 0) {
|
||||
lines.push(` Key insights: ${r.keyInsights.slice(0, 3).join(', ')}`);
|
||||
}
|
||||
parts.push('');
|
||||
}
|
||||
return lines;
|
||||
});
|
||||
|
||||
if (research.findings.codebasePatterns.length > 0) {
|
||||
parts.push('### Existing Codebase Patterns');
|
||||
for (const p of research.findings.codebasePatterns.slice(0, 5)) {
|
||||
parts.push(`- ${p.pattern} at ${p.location}`);
|
||||
}
|
||||
parts.push('');
|
||||
}
|
||||
this._formatResearchSection(parts, '### Existing Codebase Patterns', research.findings.codebasePatterns, (item) => {
|
||||
const p = item as ResearchResult['findings']['codebasePatterns'][number];
|
||||
return [`- ${p.pattern} at ${p.location}`];
|
||||
});
|
||||
|
||||
if (research.findings.technicalRecommendations.length > 0) {
|
||||
parts.push('### Recommendations');
|
||||
for (const r of research.findings.technicalRecommendations.slice(0, 5)) {
|
||||
parts.push(`- ${r}`);
|
||||
}
|
||||
parts.push('');
|
||||
}
|
||||
this._formatResearchSection(parts, '### Recommendations', research.findings.technicalRecommendations, (item) => [
|
||||
`- ${item as string}`,
|
||||
]);
|
||||
|
||||
return parts.join('\n');
|
||||
}
|
||||
@@ -414,18 +448,20 @@ export class PlanOrchestrator {
|
||||
|
||||
const durationMs = Date.now() - startTime;
|
||||
|
||||
// Extract JSON from response
|
||||
const jsonMatch = response.match(/\{[\s\S]*\}/);
|
||||
if (!jsonMatch) {
|
||||
onSubagent?.({
|
||||
type: 'failed',
|
||||
agentId,
|
||||
agentType: 'research',
|
||||
model: this.researchModel,
|
||||
status: 'failed',
|
||||
error: 'No JSON found',
|
||||
durationMs,
|
||||
});
|
||||
console.log(
|
||||
`[PlanOrchestrator] Research response length: ${response.length}, first 500 chars:`,
|
||||
response.substring(0, 500)
|
||||
);
|
||||
|
||||
// Extract JSON from response — try multiple strategies
|
||||
const jsonStr = this._extractJsonFromResponse(response);
|
||||
|
||||
if (!jsonStr) {
|
||||
console.error(
|
||||
`[PlanOrchestrator] No JSON found in research response. Full response:`,
|
||||
response.substring(0, 2000)
|
||||
);
|
||||
this._emitAgentFailure(onSubagent, agentId, 'research', this.researchModel, 'No JSON found', durationMs);
|
||||
return {
|
||||
success: false,
|
||||
findings: {
|
||||
@@ -441,17 +477,9 @@ export class PlanOrchestrator {
|
||||
};
|
||||
}
|
||||
|
||||
const parsed = tryParseJSON(jsonMatch[0]);
|
||||
const parsed = tryParseJSON(jsonStr);
|
||||
if (!parsed.success) {
|
||||
onSubagent?.({
|
||||
type: 'failed',
|
||||
agentId,
|
||||
agentType: 'research',
|
||||
model: this.researchModel,
|
||||
status: 'failed',
|
||||
error: parsed.error,
|
||||
durationMs,
|
||||
});
|
||||
this._emitAgentFailure(onSubagent, agentId, 'research', this.researchModel, parsed.error!, durationMs);
|
||||
return {
|
||||
success: false,
|
||||
findings: {
|
||||
@@ -495,16 +523,8 @@ export class PlanOrchestrator {
|
||||
return result;
|
||||
} catch (err) {
|
||||
const durationMs = Date.now() - startTime;
|
||||
const error = err instanceof Error ? err.message : String(err);
|
||||
onSubagent?.({
|
||||
type: 'failed',
|
||||
agentId,
|
||||
agentType: 'research',
|
||||
model: this.researchModel,
|
||||
status: 'failed',
|
||||
error,
|
||||
durationMs,
|
||||
});
|
||||
const error = getErrorMessage(err);
|
||||
this._emitAgentFailure(onSubagent, agentId, 'research', this.researchModel, error, durationMs);
|
||||
return {
|
||||
success: false,
|
||||
findings: {
|
||||
@@ -587,32 +607,26 @@ export class PlanOrchestrator {
|
||||
|
||||
const durationMs = Date.now() - startTime;
|
||||
|
||||
// Extract JSON from response
|
||||
const jsonMatch = response.match(/\{[\s\S]*\}/);
|
||||
if (!jsonMatch) {
|
||||
onSubagent?.({
|
||||
type: 'failed',
|
||||
agentId,
|
||||
agentType: 'planner',
|
||||
model: this.plannerModel,
|
||||
status: 'failed',
|
||||
error: 'No JSON found',
|
||||
durationMs,
|
||||
});
|
||||
console.log(
|
||||
`[PlanOrchestrator] Planner response length: ${response.length}, first 500 chars:`,
|
||||
response.substring(0, 500)
|
||||
);
|
||||
|
||||
// Extract JSON from response — try multiple strategies
|
||||
const jsonStr = this._extractJsonFromResponse(response);
|
||||
|
||||
if (!jsonStr) {
|
||||
console.error(
|
||||
`[PlanOrchestrator] No JSON found in planner response. Full response:`,
|
||||
response.substring(0, 2000)
|
||||
);
|
||||
this._emitAgentFailure(onSubagent, agentId, 'planner', this.plannerModel, 'No JSON found', durationMs);
|
||||
return { success: false, error: 'No JSON in response' };
|
||||
}
|
||||
|
||||
const parsed = tryParseJSON(jsonMatch[0]);
|
||||
const parsed = tryParseJSON(jsonStr);
|
||||
if (!parsed.success) {
|
||||
onSubagent?.({
|
||||
type: 'failed',
|
||||
agentId,
|
||||
agentType: 'planner',
|
||||
model: this.plannerModel,
|
||||
status: 'failed',
|
||||
error: parsed.error,
|
||||
durationMs,
|
||||
});
|
||||
this._emitAgentFailure(onSubagent, agentId, 'planner', this.plannerModel, parsed.error!, durationMs);
|
||||
return { success: false, error: parsed.error };
|
||||
}
|
||||
|
||||
@@ -637,16 +651,8 @@ export class PlanOrchestrator {
|
||||
return { success: true, items, gaps, warnings };
|
||||
} catch (err) {
|
||||
const durationMs = Date.now() - startTime;
|
||||
const error = err instanceof Error ? err.message : String(err);
|
||||
onSubagent?.({
|
||||
type: 'failed',
|
||||
agentId,
|
||||
agentType: 'planner',
|
||||
model: this.plannerModel,
|
||||
status: 'failed',
|
||||
error,
|
||||
durationMs,
|
||||
});
|
||||
const error = getErrorMessage(err);
|
||||
this._emitAgentFailure(onSubagent, agentId, 'planner', this.plannerModel, error, durationMs);
|
||||
return { success: false, error };
|
||||
} finally {
|
||||
// Always clean up session and progress interval — centralizing here
|
||||
|
||||
@@ -7,3 +7,4 @@
|
||||
|
||||
export { RESEARCH_AGENT_PROMPT } from './research-agent.js';
|
||||
export { PLANNER_PROMPT } from './planner.js';
|
||||
export { PHASE_EXECUTION_PROMPT, TEAM_LEAD_PROMPT, REPLAN_PROMPT, SINGLE_TASK_PROMPT } from './orchestrator.js';
|
||||
|
||||
@@ -0,0 +1,100 @@
|
||||
/**
|
||||
* @fileoverview Orchestrator Loop prompt templates.
|
||||
*
|
||||
* Templates for phase execution, team delegation, verification, and replanning.
|
||||
* Placeholders use {VARIABLE} syntax and are replaced at runtime.
|
||||
*
|
||||
* @module prompts/orchestrator
|
||||
*/
|
||||
|
||||
/**
|
||||
* Phase execution prompt — tells Claude what to accomplish in this phase.
|
||||
*
|
||||
* Placeholders:
|
||||
* - {PHASE_NUMBER}: Phase index (1-based)
|
||||
* - {PHASE_NAME}: Human-readable phase name
|
||||
* - {GOAL}: Original user goal
|
||||
* - {COMPLETED_PHASES}: Summary of previously completed phases
|
||||
* - {TASK_LIST}: Numbered task list for this phase
|
||||
* - {VERIFICATION_CRITERIA}: What will be checked after this phase
|
||||
* - {COMPLETION_PHRASE}: The phrase to output when done
|
||||
*/
|
||||
export const PHASE_EXECUTION_PROMPT = `You are executing {PHASE_NAME} of a larger project.
|
||||
|
||||
OVERALL GOAL: {GOAL}
|
||||
|
||||
COMPLETED SO FAR:
|
||||
{COMPLETED_PHASES}
|
||||
|
||||
YOUR TASKS FOR THIS PHASE:
|
||||
{TASK_LIST}
|
||||
|
||||
Complete each task thoroughly. Run tests after each change to catch issues early.
|
||||
|
||||
VERIFICATION (will be checked after you finish):
|
||||
{VERIFICATION_CRITERIA}
|
||||
|
||||
When ALL tasks in this phase are complete and verified, output: <promise>{COMPLETION_PHRASE}</promise>`;
|
||||
|
||||
/**
|
||||
* Team lead delegation prompt — instructs a lead to coordinate teammates.
|
||||
*
|
||||
* Placeholders:
|
||||
* - {PHASE_NAME}: Phase name
|
||||
* - {TASK_LIST}: Numbered task list
|
||||
* - {TEAMMATE_HINTS}: Suggested teammate specializations
|
||||
* - {COMPLETION_PHRASE}: Phrase for when all work is done
|
||||
*/
|
||||
export const TEAM_LEAD_PROMPT = `You are the team lead for {PHASE_NAME}.
|
||||
|
||||
Create teammates and delegate the following tasks for parallel execution:
|
||||
|
||||
{TASK_LIST}
|
||||
|
||||
Suggested teammate roles:
|
||||
{TEAMMATE_HINTS}
|
||||
|
||||
Each teammate should focus on their assigned task area. Monitor their progress.
|
||||
When ALL tasks are complete and you've verified the results, output: <promise>{COMPLETION_PHRASE}</promise>`;
|
||||
|
||||
/**
|
||||
* Replan prompt — gives failure context and asks for recovery.
|
||||
*
|
||||
* Placeholders:
|
||||
* - {PHASE_NAME}: Phase name
|
||||
* - {ATTEMPT_NUMBER}: Current retry attempt
|
||||
* - {MAX_ATTEMPTS}: Maximum attempts allowed
|
||||
* - {FAILURE_SUMMARY}: What went wrong
|
||||
* - {SUGGESTIONS}: Recovery suggestions from verification
|
||||
* - {ORIGINAL_TASKS}: The original task list
|
||||
* - {COMPLETION_PHRASE}: Phrase for when recovery is done
|
||||
*/
|
||||
export const REPLAN_PROMPT = `Phase "{PHASE_NAME}" verification failed (attempt {ATTEMPT_NUMBER}/{MAX_ATTEMPTS}).
|
||||
|
||||
WHAT WENT WRONG:
|
||||
{FAILURE_SUMMARY}
|
||||
|
||||
SUGGESTIONS:
|
||||
{SUGGESTIONS}
|
||||
|
||||
ORIGINAL TASKS:
|
||||
{ORIGINAL_TASKS}
|
||||
|
||||
Fix the issues identified above. Focus on making the verification criteria pass.
|
||||
When the fixes are complete, output: <promise>{COMPLETION_PHRASE}</promise>`;
|
||||
|
||||
/**
|
||||
* Single-task execution prompt — for phases with a single task.
|
||||
*
|
||||
* Placeholders:
|
||||
* - {TASK}: The task description
|
||||
* - {GOAL}: Original user goal
|
||||
* - {CONTEXT}: Any relevant context
|
||||
* - {COMPLETION_PHRASE}: Phrase for when done
|
||||
*/
|
||||
export const SINGLE_TASK_PROMPT = `{TASK}
|
||||
|
||||
Context: This is part of a larger project — {GOAL}
|
||||
{CONTEXT}
|
||||
|
||||
When done, output: <promise>{COMPLETION_PHRASE}</promise>`;
|
||||
@@ -8,12 +8,12 @@
|
||||
|
||||
import { existsSync, readFileSync, writeFileSync, mkdirSync } from 'node:fs';
|
||||
import { join } from 'node:path';
|
||||
import { homedir } from 'node:os';
|
||||
import webpush from 'web-push';
|
||||
import type { VapidKeys, PushSubscriptionRecord } from './types.js';
|
||||
import { Debouncer } from './utils/index.js';
|
||||
import { getDataDir } from './config/instance.js';
|
||||
|
||||
const DATA_DIR = join(homedir(), '.codeman');
|
||||
const DATA_DIR = getDataDir();
|
||||
const KEYS_FILE = join(DATA_DIR, 'push-keys.json');
|
||||
const SUBS_FILE = join(DATA_DIR, 'push-subscriptions.json');
|
||||
const SAVE_DEBOUNCE_MS = 500;
|
||||
|
||||
@@ -24,7 +24,7 @@ const YAML_LINE_PATTERN = /^([a-zA-Z_-]+):\s*"?([^"\n]+)"?\s*$/gm;
|
||||
/**
|
||||
* Ralph Loop configuration from .claude/ralph-loop.local.md
|
||||
*/
|
||||
export interface RalphLoopConfig {
|
||||
interface RalphLoopConfig {
|
||||
enabled: boolean;
|
||||
iteration: number;
|
||||
maxIterations: number | null;
|
||||
|
||||
@@ -34,19 +34,11 @@ import { RalphLoopStatus, getErrorMessage } from './types.js';
|
||||
/**
|
||||
* Events emitted by RalphLoop
|
||||
*/
|
||||
export interface RalphLoopEvents {
|
||||
started: () => void;
|
||||
stopped: () => void;
|
||||
taskAssigned: (taskId: string, sessionId: string) => void;
|
||||
taskCompleted: (taskId: string) => void;
|
||||
taskFailed: (taskId: string, error: string) => void;
|
||||
error: (error: Error) => void;
|
||||
}
|
||||
|
||||
/**
|
||||
* Configuration options for RalphLoop
|
||||
*/
|
||||
export interface RalphLoopOptions {
|
||||
interface RalphLoopOptions {
|
||||
/** How often to check for new tasks (default from config) */
|
||||
pollIntervalMs?: number;
|
||||
/** Minimum time to run before stopping (null = no minimum) */
|
||||
|
||||
@@ -91,6 +91,67 @@ const COMPLETION_INDICATOR_PATTERNS = [
|
||||
/project\s+(?:is\s+)?(?:completed?|done|finished)/i,
|
||||
];
|
||||
|
||||
interface FieldParser<T> {
|
||||
pattern: RegExp;
|
||||
field: keyof RalphStatusBlock;
|
||||
validate: (value: string) => boolean;
|
||||
transform: (value: string) => T;
|
||||
errorMsg: (value: string) => string;
|
||||
}
|
||||
|
||||
const FIELD_PARSERS: FieldParser<RalphStatusValue | RalphTestsStatus | RalphWorkType | number | boolean | string>[] = [
|
||||
{
|
||||
pattern: RALPH_STATUS_FIELD_PATTERN,
|
||||
field: 'status',
|
||||
validate: (v) => ['IN_PROGRESS', 'COMPLETE', 'BLOCKED'].includes(v.toUpperCase()),
|
||||
transform: (v) => v.toUpperCase() as RalphStatusValue,
|
||||
errorMsg: (v) => `Invalid STATUS value: "${v}". Expected: IN_PROGRESS, COMPLETE, or BLOCKED`,
|
||||
},
|
||||
{
|
||||
pattern: RALPH_TASKS_COMPLETED_PATTERN,
|
||||
field: 'tasksCompletedThisLoop',
|
||||
validate: (v) => !Number.isNaN(parseInt(v, 10)) && parseInt(v, 10) >= 0,
|
||||
transform: (v) => parseInt(v, 10),
|
||||
errorMsg: (v) => `Invalid TASKS_COMPLETED_THIS_LOOP value: "${v}". Expected: non-negative integer`,
|
||||
},
|
||||
{
|
||||
pattern: RALPH_FILES_MODIFIED_PATTERN,
|
||||
field: 'filesModified',
|
||||
validate: (v) => !Number.isNaN(parseInt(v, 10)) && parseInt(v, 10) >= 0,
|
||||
transform: (v) => parseInt(v, 10),
|
||||
errorMsg: (v) => `Invalid FILES_MODIFIED value: "${v}". Expected: non-negative integer`,
|
||||
},
|
||||
{
|
||||
pattern: RALPH_TESTS_STATUS_PATTERN,
|
||||
field: 'testsStatus',
|
||||
validate: (v) => ['PASSING', 'FAILING', 'NOT_RUN'].includes(v.toUpperCase()),
|
||||
transform: (v) => v.toUpperCase() as RalphTestsStatus,
|
||||
errorMsg: (v) => `Invalid TESTS_STATUS value: "${v}". Expected: PASSING, FAILING, or NOT_RUN`,
|
||||
},
|
||||
{
|
||||
pattern: RALPH_WORK_TYPE_PATTERN,
|
||||
field: 'workType',
|
||||
validate: (v) => ['IMPLEMENTATION', 'TESTING', 'DOCUMENTATION', 'REFACTORING'].includes(v.toUpperCase()),
|
||||
transform: (v) => v.toUpperCase() as RalphWorkType,
|
||||
errorMsg: (v) =>
|
||||
`Invalid WORK_TYPE value: "${v}". Expected: IMPLEMENTATION, TESTING, DOCUMENTATION, or REFACTORING`,
|
||||
},
|
||||
{
|
||||
pattern: RALPH_EXIT_SIGNAL_PATTERN,
|
||||
field: 'exitSignal',
|
||||
validate: () => true,
|
||||
transform: (v) => v.toLowerCase() === 'true',
|
||||
errorMsg: () => '',
|
||||
},
|
||||
{
|
||||
pattern: RALPH_RECOMMENDATION_PATTERN,
|
||||
field: 'recommendation',
|
||||
validate: () => true,
|
||||
transform: (v) => v.trim(),
|
||||
errorMsg: () => '',
|
||||
},
|
||||
];
|
||||
|
||||
/**
|
||||
* RalphStatusParser - Parses RALPH_STATUS blocks and manages circuit breaker.
|
||||
*
|
||||
@@ -303,85 +364,21 @@ export class RalphStatusParser extends EventEmitter {
|
||||
const trimmedLine = line.trim();
|
||||
if (!trimmedLine) continue;
|
||||
|
||||
// Track whether this line matched any known field
|
||||
let matched = false;
|
||||
|
||||
// STATUS field (required)
|
||||
const statusMatch = trimmedLine.match(RALPH_STATUS_FIELD_PATTERN);
|
||||
if (statusMatch) {
|
||||
const value = statusMatch[1].toUpperCase();
|
||||
if (['IN_PROGRESS', 'COMPLETE', 'BLOCKED'].includes(value)) {
|
||||
block.status = value as RalphStatusValue;
|
||||
} else {
|
||||
parseErrors.push(`Invalid STATUS value: "${value}". Expected: IN_PROGRESS, COMPLETE, or BLOCKED`);
|
||||
for (const parser of FIELD_PARSERS) {
|
||||
const match = trimmedLine.match(parser.pattern);
|
||||
if (match) {
|
||||
const rawValue = match[1];
|
||||
if (parser.validate(rawValue)) {
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
(block as any)[parser.field] = parser.transform(rawValue);
|
||||
} else {
|
||||
parseErrors.push(parser.errorMsg(rawValue));
|
||||
}
|
||||
matched = true;
|
||||
break;
|
||||
}
|
||||
matched = true;
|
||||
}
|
||||
|
||||
// TASKS_COMPLETED_THIS_LOOP field
|
||||
const tasksMatch = trimmedLine.match(RALPH_TASKS_COMPLETED_PATTERN);
|
||||
if (tasksMatch) {
|
||||
const value = parseInt(tasksMatch[1], 10);
|
||||
if (!Number.isNaN(value) && value >= 0) {
|
||||
block.tasksCompletedThisLoop = value;
|
||||
} else {
|
||||
parseErrors.push(
|
||||
`Invalid TASKS_COMPLETED_THIS_LOOP value: "${tasksMatch[1]}". Expected: non-negative integer`
|
||||
);
|
||||
}
|
||||
matched = true;
|
||||
}
|
||||
|
||||
// FILES_MODIFIED field
|
||||
const filesMatch = trimmedLine.match(RALPH_FILES_MODIFIED_PATTERN);
|
||||
if (filesMatch) {
|
||||
const value = parseInt(filesMatch[1], 10);
|
||||
if (!Number.isNaN(value) && value >= 0) {
|
||||
block.filesModified = value;
|
||||
} else {
|
||||
parseErrors.push(`Invalid FILES_MODIFIED value: "${filesMatch[1]}". Expected: non-negative integer`);
|
||||
}
|
||||
matched = true;
|
||||
}
|
||||
|
||||
// TESTS_STATUS field
|
||||
const testsMatch = trimmedLine.match(RALPH_TESTS_STATUS_PATTERN);
|
||||
if (testsMatch) {
|
||||
const value = testsMatch[1].toUpperCase();
|
||||
if (['PASSING', 'FAILING', 'NOT_RUN'].includes(value)) {
|
||||
block.testsStatus = value as RalphTestsStatus;
|
||||
} else {
|
||||
parseErrors.push(`Invalid TESTS_STATUS value: "${value}". Expected: PASSING, FAILING, or NOT_RUN`);
|
||||
}
|
||||
matched = true;
|
||||
}
|
||||
|
||||
// WORK_TYPE field
|
||||
const workMatch = trimmedLine.match(RALPH_WORK_TYPE_PATTERN);
|
||||
if (workMatch) {
|
||||
const value = workMatch[1].toUpperCase();
|
||||
if (['IMPLEMENTATION', 'TESTING', 'DOCUMENTATION', 'REFACTORING'].includes(value)) {
|
||||
block.workType = value as RalphWorkType;
|
||||
} else {
|
||||
parseErrors.push(
|
||||
`Invalid WORK_TYPE value: "${value}". Expected: IMPLEMENTATION, TESTING, DOCUMENTATION, or REFACTORING`
|
||||
);
|
||||
}
|
||||
matched = true;
|
||||
}
|
||||
|
||||
// EXIT_SIGNAL field
|
||||
const exitMatch = trimmedLine.match(RALPH_EXIT_SIGNAL_PATTERN);
|
||||
if (exitMatch) {
|
||||
block.exitSignal = exitMatch[1].toLowerCase() === 'true';
|
||||
matched = true;
|
||||
}
|
||||
|
||||
// RECOMMENDATION field
|
||||
const recMatch = trimmedLine.match(RALPH_RECOMMENDATION_PATTERN);
|
||||
if (recMatch) {
|
||||
block.recommendation = recMatch[1].trim();
|
||||
matched = true;
|
||||
}
|
||||
|
||||
// Track unknown fields for debugging (only if looks like a field)
|
||||
@@ -475,38 +472,9 @@ export class RalphStatusParser extends EventEmitter {
|
||||
const prevState = this._circuitBreaker.state;
|
||||
|
||||
if (hasProgress) {
|
||||
// Progress detected - reset counters, possibly close circuit
|
||||
this._circuitBreaker.consecutiveNoProgress = 0;
|
||||
this._circuitBreaker.consecutiveSameError = 0;
|
||||
this._circuitBreaker.lastProgressIteration = this._cycleCount;
|
||||
|
||||
if (this._circuitBreaker.state === 'HALF_OPEN') {
|
||||
this._circuitBreaker.state = 'CLOSED';
|
||||
this._circuitBreaker.reason = 'Progress detected, circuit closed';
|
||||
this._circuitBreaker.reasonCode = 'progress_detected';
|
||||
}
|
||||
this._handleProgressDetected();
|
||||
} else {
|
||||
// No progress
|
||||
this._circuitBreaker.consecutiveNoProgress++;
|
||||
|
||||
// State transitions based on consecutive no-progress
|
||||
if (this._circuitBreaker.state === 'CLOSED') {
|
||||
if (this._circuitBreaker.consecutiveNoProgress >= 3) {
|
||||
this._circuitBreaker.state = 'OPEN';
|
||||
this._circuitBreaker.reason = `No progress for ${this._circuitBreaker.consecutiveNoProgress} iterations`;
|
||||
this._circuitBreaker.reasonCode = 'no_progress_open';
|
||||
} else if (this._circuitBreaker.consecutiveNoProgress >= 2) {
|
||||
this._circuitBreaker.state = 'HALF_OPEN';
|
||||
this._circuitBreaker.reason = 'Warning: no progress detected';
|
||||
this._circuitBreaker.reasonCode = 'no_progress_warning';
|
||||
}
|
||||
} else if (this._circuitBreaker.state === 'HALF_OPEN') {
|
||||
if (this._circuitBreaker.consecutiveNoProgress >= 3) {
|
||||
this._circuitBreaker.state = 'OPEN';
|
||||
this._circuitBreaker.reason = `No progress for ${this._circuitBreaker.consecutiveNoProgress} iterations`;
|
||||
this._circuitBreaker.reasonCode = 'no_progress_open';
|
||||
}
|
||||
}
|
||||
this._handleNoProgress();
|
||||
}
|
||||
|
||||
// Track tests failure
|
||||
@@ -535,6 +503,40 @@ export class RalphStatusParser extends EventEmitter {
|
||||
}
|
||||
}
|
||||
|
||||
private _handleProgressDetected(): void {
|
||||
this._circuitBreaker.consecutiveNoProgress = 0;
|
||||
this._circuitBreaker.consecutiveSameError = 0;
|
||||
this._circuitBreaker.lastProgressIteration = this._cycleCount;
|
||||
|
||||
if (this._circuitBreaker.state === 'HALF_OPEN') {
|
||||
this._circuitBreaker.state = 'CLOSED';
|
||||
this._circuitBreaker.reason = 'Progress detected, circuit closed';
|
||||
this._circuitBreaker.reasonCode = 'progress_detected';
|
||||
}
|
||||
}
|
||||
|
||||
private _handleNoProgress(): void {
|
||||
this._circuitBreaker.consecutiveNoProgress++;
|
||||
|
||||
if (this._circuitBreaker.state === 'CLOSED') {
|
||||
if (this._circuitBreaker.consecutiveNoProgress >= 3) {
|
||||
this._circuitBreaker.state = 'OPEN';
|
||||
this._circuitBreaker.reason = `No progress for ${this._circuitBreaker.consecutiveNoProgress} iterations`;
|
||||
this._circuitBreaker.reasonCode = 'no_progress_open';
|
||||
} else if (this._circuitBreaker.consecutiveNoProgress >= 2) {
|
||||
this._circuitBreaker.state = 'HALF_OPEN';
|
||||
this._circuitBreaker.reason = 'Warning: no progress detected';
|
||||
this._circuitBreaker.reasonCode = 'no_progress_warning';
|
||||
}
|
||||
} else if (this._circuitBreaker.state === 'HALF_OPEN') {
|
||||
if (this._circuitBreaker.consecutiveNoProgress >= 3) {
|
||||
this._circuitBreaker.state = 'OPEN';
|
||||
this._circuitBreaker.reason = `No progress for ${this._circuitBreaker.consecutiveNoProgress} iterations`;
|
||||
this._circuitBreaker.reasonCode = 'no_progress_open';
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Check line for completion indicators (natural language patterns).
|
||||
* Used for dual-condition exit gate.
|
||||
|
||||
@@ -19,7 +19,6 @@
|
||||
* Key exports:
|
||||
* - `RalphTracker` class — main tracker, extends EventEmitter
|
||||
* - `RalphTrackerEvents` interface — typed event map
|
||||
* - Re-exports: `EnhancedPlanTask`, `CheckpointReview` from ralph-plan-tracker
|
||||
*
|
||||
* Key methods: `processData(data)` — feed terminal output, `getState()`,
|
||||
* `getTodos()`, `getCompletionHistory()`, `getPlanTasks()`, `reset()`
|
||||
@@ -66,9 +65,6 @@ import { RalphStallDetector } from './ralph-stall-detector.js';
|
||||
import { RalphStatusParser } from './ralph-status-parser.js';
|
||||
import { STALE_DATA_MAX_AGE_MS, INACTIVITY_TIMEOUT_MS } from './config/server-timing.js';
|
||||
|
||||
// Re-export sub-module types for backward compatibility
|
||||
export type { EnhancedPlanTask, CheckpointReview } from './ralph-plan-tracker.js';
|
||||
|
||||
// ========== Configuration Constants ==========
|
||||
// Note: MAX_TODOS_PER_SESSION and MAX_LINE_BUFFER_SIZE are imported from config modules
|
||||
|
||||
@@ -100,6 +96,18 @@ const TODO_CLEANUP_INTERVAL_MS = INACTIVITY_TIMEOUT_MS;
|
||||
*/
|
||||
const TODO_SIMILARITY_THRESHOLD = 0.85;
|
||||
|
||||
/**
|
||||
* Similarity threshold for short todo content (<30 chars).
|
||||
* Higher threshold reduces false positive deduplication of short strings.
|
||||
*/
|
||||
const SIMILARITY_THRESHOLD_SHORT = 0.95;
|
||||
|
||||
/**
|
||||
* Similarity threshold for medium-length todo content (30-60 chars).
|
||||
* Slightly relaxed compared to short strings.
|
||||
*/
|
||||
const SIMILARITY_THRESHOLD_MEDIUM = 0.9;
|
||||
|
||||
/**
|
||||
* Debounce interval for event emissions (milliseconds).
|
||||
* Prevents UI jitter from rapid consecutive updates.
|
||||
@@ -378,32 +386,6 @@ const P2_PRIORITY_PATTERNS = [
|
||||
* @event circuitBreakerUpdate - Fired when circuit breaker state changes
|
||||
* @event exitGateMet - Fired when dual-condition exit gate is met
|
||||
*/
|
||||
export interface RalphTrackerEvents {
|
||||
/** Emitted when loop state changes */
|
||||
loopUpdate: (state: RalphTrackerState) => void;
|
||||
/** Emitted when todo list is modified */
|
||||
todoUpdate: (todos: RalphTodoItem[]) => void;
|
||||
/** Emitted when completion phrase detected (loop finished) */
|
||||
completionDetected: (phrase: string) => void;
|
||||
/** Emitted when tracker auto-enables from disabled state */
|
||||
enabled: () => void;
|
||||
/** Emitted when a RALPH_STATUS block is parsed */
|
||||
statusBlockDetected: (block: RalphStatusBlock) => void;
|
||||
/** Emitted when circuit breaker state changes */
|
||||
circuitBreakerUpdate: (status: CircuitBreakerStatus) => void;
|
||||
/** Emitted when dual-condition exit gate is met (completion indicators >= 2 AND EXIT_SIGNAL: true) */
|
||||
exitGateMet: (data: { completionIndicators: number; exitSignal: boolean }) => void;
|
||||
/** Emitted when iteration count hasn't changed for an extended period (stall warning) */
|
||||
iterationStallWarning: (data: { iteration: number; stallDurationMs: number }) => void;
|
||||
/** Emitted when iteration count hasn't changed for critical period (stall critical) */
|
||||
iterationStallCritical: (data: { iteration: number; stallDurationMs: number }) => void;
|
||||
/** Emitted when a common/risky completion phrase is detected (P1-002) */
|
||||
phraseValidationWarning: (data: {
|
||||
phrase: string;
|
||||
reason: 'common' | 'short' | 'numeric';
|
||||
suggestedPhrase: string;
|
||||
}) => void;
|
||||
}
|
||||
|
||||
/**
|
||||
* RalphTracker - Parses terminal output to detect Ralph Wiggum loops and todos
|
||||
@@ -1299,6 +1281,24 @@ export class RalphTracker extends EventEmitter {
|
||||
this.detectTodoItems(trimmed);
|
||||
}
|
||||
|
||||
/**
|
||||
* Mark all tracked todos as completed and emit todoUpdate if any changed.
|
||||
* @returns true if any todo was updated
|
||||
*/
|
||||
private completeAllTodos(): boolean {
|
||||
let updated = false;
|
||||
for (const todo of this._todos.values()) {
|
||||
if (todo.status !== 'completed') {
|
||||
todo.status = 'completed';
|
||||
updated = true;
|
||||
}
|
||||
}
|
||||
if (updated) {
|
||||
this.emit('todoUpdate', this.todos);
|
||||
}
|
||||
return updated;
|
||||
}
|
||||
|
||||
/**
|
||||
* Detect "all tasks complete" messages.
|
||||
*/
|
||||
@@ -1318,16 +1318,7 @@ export class RalphTracker extends EventEmitter {
|
||||
return;
|
||||
}
|
||||
|
||||
let updated = false;
|
||||
for (const todo of this._todos.values()) {
|
||||
if (todo.status !== 'completed') {
|
||||
todo.status = 'completed';
|
||||
updated = true;
|
||||
}
|
||||
}
|
||||
if (updated) {
|
||||
this.emit('todoUpdate', this.todos);
|
||||
}
|
||||
this.completeAllTodos();
|
||||
|
||||
if (this._loopState.completionPhrase) {
|
||||
this._loopState.active = false;
|
||||
@@ -1425,16 +1416,7 @@ export class RalphTracker extends EventEmitter {
|
||||
|
||||
if (bareCount > 1) return;
|
||||
|
||||
let updated = false;
|
||||
for (const todo of this._todos.values()) {
|
||||
if (todo.status !== 'completed') {
|
||||
todo.status = 'completed';
|
||||
updated = true;
|
||||
}
|
||||
}
|
||||
if (updated) {
|
||||
this.emit('todoUpdate', this.todos);
|
||||
}
|
||||
this.completeAllTodos();
|
||||
|
||||
this._loopState.active = false;
|
||||
this._loopState.lastActivity = Date.now();
|
||||
@@ -1480,16 +1462,7 @@ export class RalphTracker extends EventEmitter {
|
||||
if (canonicalCount >= 2 || this._loopState.active) {
|
||||
this._loopState.active = false;
|
||||
this._loopState.lastActivity = Date.now();
|
||||
let updated = false;
|
||||
for (const todo of this._todos.values()) {
|
||||
if (todo.status !== 'completed') {
|
||||
todo.status = 'completed';
|
||||
updated = true;
|
||||
}
|
||||
}
|
||||
if (updated) {
|
||||
this.emit('todoUpdate', this.todos);
|
||||
}
|
||||
this.completeAllTodos();
|
||||
this.emit('completionDetected', matchedPhrase);
|
||||
this.emit('loopUpdate', this.loopState);
|
||||
return;
|
||||
@@ -1497,16 +1470,7 @@ export class RalphTracker extends EventEmitter {
|
||||
}
|
||||
|
||||
if (this._loopState.active || count >= 2) {
|
||||
let updated = false;
|
||||
for (const todo of this._todos.values()) {
|
||||
if (todo.status !== 'completed') {
|
||||
todo.status = 'completed';
|
||||
updated = true;
|
||||
}
|
||||
}
|
||||
if (updated) {
|
||||
this.emit('todoUpdate', this.todos);
|
||||
}
|
||||
this.completeAllTodos();
|
||||
|
||||
this._loopState.active = false;
|
||||
this._loopState.lastActivity = Date.now();
|
||||
@@ -1532,41 +1496,39 @@ export class RalphTracker extends EventEmitter {
|
||||
const suggestedPhrase = `${phrase}_${uniqueSuffix}`;
|
||||
|
||||
if (COMMON_COMPLETION_PHRASES.has(normalized)) {
|
||||
console.warn(
|
||||
`[RalphTracker] Warning: Completion phrase "${phrase}" is very common and may cause false positives. Consider using: "${suggestedPhrase}"`
|
||||
);
|
||||
this.emit('phraseValidationWarning', {
|
||||
phrase,
|
||||
reason: 'common',
|
||||
suggestedPhrase,
|
||||
});
|
||||
this.emitValidationWarning(phrase, 'common', suggestedPhrase);
|
||||
return;
|
||||
}
|
||||
|
||||
if (normalized.length < MIN_RECOMMENDED_PHRASE_LENGTH) {
|
||||
console.warn(
|
||||
`[RalphTracker] Warning: Completion phrase "${phrase}" is too short (${normalized.length} chars). Consider using: "${suggestedPhrase}"`
|
||||
);
|
||||
this.emit('phraseValidationWarning', {
|
||||
phrase,
|
||||
reason: 'short',
|
||||
suggestedPhrase,
|
||||
});
|
||||
this.emitValidationWarning(phrase, 'short', suggestedPhrase);
|
||||
return;
|
||||
}
|
||||
|
||||
if (/^\d+$/.test(normalized)) {
|
||||
console.warn(
|
||||
`[RalphTracker] Warning: Completion phrase "${phrase}" is numeric-only and may cause false positives. Consider using: "${suggestedPhrase}"`
|
||||
);
|
||||
this.emit('phraseValidationWarning', {
|
||||
phrase,
|
||||
reason: 'numeric',
|
||||
suggestedPhrase,
|
||||
});
|
||||
this.emitValidationWarning(phrase, 'numeric', suggestedPhrase);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Emit a phrase validation warning with a console message and event.
|
||||
*/
|
||||
private emitValidationWarning(phrase: string, reason: 'common' | 'short' | 'numeric', suggestedPhrase: string): void {
|
||||
const descriptions: Record<'common' | 'short' | 'numeric', string> = {
|
||||
common: 'is very common and may cause false positives',
|
||||
short: `is too short (${phrase.toUpperCase().replace(/[\s_\-.]+/g, '').length} chars)`,
|
||||
numeric: 'is numeric-only and may cause false positives',
|
||||
};
|
||||
console.warn(
|
||||
`[RalphTracker] Warning: Completion phrase "${phrase}" ${descriptions[reason]}. Consider using: "${suggestedPhrase}"`
|
||||
);
|
||||
this.emit('phraseValidationWarning', {
|
||||
phrase,
|
||||
reason,
|
||||
suggestedPhrase,
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Activate the loop if not already active.
|
||||
*/
|
||||
@@ -1977,9 +1939,9 @@ export class RalphTracker extends EventEmitter {
|
||||
|
||||
let threshold: number;
|
||||
if (normalized.length < 30) {
|
||||
threshold = 0.95;
|
||||
threshold = SIMILARITY_THRESHOLD_SHORT;
|
||||
} else if (normalized.length < 60) {
|
||||
threshold = 0.9;
|
||||
threshold = SIMILARITY_THRESHOLD_MEDIUM;
|
||||
} else {
|
||||
threshold = TODO_SIMILARITY_THRESHOLD;
|
||||
}
|
||||
|
||||