mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-09-30 20:49:41 +02:00
Compare commits
266
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
848ab48b0a | ||
|
|
0b106b03eb | ||
|
|
54c591c84d | ||
|
|
2eece4f8f9 | ||
|
|
4d165d3fb1 | ||
|
|
d5ffc22f4a | ||
|
|
c2dfc775a3 | ||
|
|
e439cf0ef3 | ||
|
|
1f4c390e12 | ||
|
|
714050fe8a | ||
|
|
dfd3df8289 | ||
|
|
a0fbd1d28d | ||
|
|
272b56d47b | ||
|
|
1645ef5f5c | ||
|
|
627b76739c | ||
|
|
83e39c40a1 | ||
|
|
fec0409315 | ||
|
|
b4954c14cd | ||
|
|
c9f47b095a | ||
|
|
47ac16d6ab | ||
|
|
6d147c1bf1 | ||
|
|
92921b9107 | ||
|
|
614c7e6cd5 | ||
|
|
7659ca8b44 | ||
|
|
61037082d1 | ||
|
|
e71971cab4 | ||
|
|
e60b5a8a2c | ||
|
|
1d85909a06 | ||
|
|
1da2fa2529 | ||
|
|
45ea2e1d32 | ||
|
|
f6aa50239f | ||
|
|
9676e90133 | ||
|
|
bf73a84732 | ||
|
|
8d358aaa26 | ||
|
|
a9b48320a3 | ||
|
|
95a3b87062 | ||
|
|
8cef31086b | ||
|
|
55790964b7 | ||
|
|
5ae574374f | ||
|
|
47f209bf0a | ||
|
|
d81a4a76de | ||
|
|
8841bcc93f | ||
|
|
77ba41f8da | ||
|
|
b80d47aff8 | ||
|
|
d67da5c9d0 | ||
|
|
d6c3386102 | ||
|
|
10c263a5b8 | ||
|
|
b070c9ee65 | ||
|
|
e98127a804 | ||
|
|
3e3a4612e6 | ||
|
|
a5283c565d | ||
|
|
b46588f247 | ||
|
|
e6ddb0485a | ||
|
|
334884e96a | ||
|
|
69a71287e6 | ||
|
|
7485afecaf | ||
|
|
0a52a99ca9 | ||
|
|
c46e87fd7a | ||
|
|
dd230b0b6e | ||
|
|
120d780267 | ||
|
|
0af925fe82 | ||
|
|
b404dacfde | ||
|
|
cbd1fa639d | ||
|
|
43d4be8eeb | ||
|
|
f4d1ee8027 | ||
|
|
7a30a31430 | ||
|
|
fdfcc15c10 | ||
|
|
697b05b118 | ||
|
|
0462a5d5a0 | ||
|
|
da6fa663e7 | ||
|
|
10f87428c3 | ||
|
|
13e652e43f | ||
|
|
2afb1c2c2e | ||
|
|
2fb744f865 | ||
|
|
de4b1db490 | ||
|
|
8536aaef7b | ||
|
|
94b093b617 | ||
|
|
6a01412af9 | ||
|
|
bc04b6457e | ||
|
|
5b5e932ec4 | ||
|
|
f1dfbcdd65 | ||
|
|
233af33dac | ||
|
|
993e5e021c | ||
|
|
02e40f506b | ||
|
|
e558264977 | ||
|
|
9a48c43aa1 | ||
|
|
5cf5a45438 | ||
|
|
110c4696ad | ||
|
|
ac6236b268 | ||
|
|
00f022ccf8 | ||
|
|
b13672596f | ||
|
|
8d45b92eba | ||
|
|
05c788ce9d | ||
|
|
9c286eeddf | ||
|
|
e1e7dc5bd8 | ||
|
|
ce80b7a212 | ||
|
|
e587d84590 | ||
|
|
d9fa9ba1eb | ||
|
|
abf1d1f1ca | ||
|
|
90fd0a5a15 | ||
|
|
64c288a683 | ||
|
|
abd39318e6 | ||
|
|
c0422c4e21 | ||
|
|
74884a20eb | ||
|
|
1cb0441bd8 | ||
|
|
3f2cde2db7 | ||
|
|
47e7935274 | ||
|
|
a2dcc91ddf | ||
|
|
c67c130caa | ||
|
|
90a95f562b | ||
|
|
02dc46dcd7 | ||
|
|
5b4878df3b | ||
|
|
3b714446b4 | ||
|
|
9ba90a674a | ||
|
|
d3ee9f23c2 | ||
|
|
9466acfc1a | ||
|
|
e899af4305 | ||
|
|
299a21d5f5 | ||
|
|
d8e85285c9 | ||
|
|
0f955327b2 | ||
|
|
d3f2ec0220 | ||
|
|
6ef71ec3b9 | ||
|
|
d47f93abdb | ||
|
|
dcf9437308 | ||
|
|
aa13af1f7f | ||
|
|
9a2e14a93a | ||
|
|
ecb95b5d67 | ||
|
|
a7452dc046 | ||
|
|
72d437ab63 | ||
|
|
1ba0684438 | ||
|
|
af744bdb54 | ||
|
|
46d8b92049 | ||
|
|
0b3e086334 | ||
|
|
fafef0aa00 | ||
|
|
3152ec801d | ||
|
|
2e3e245cc6 | ||
|
|
165cfb52d6 | ||
|
|
6c8bd6c606 | ||
|
|
8d3bde5469 | ||
|
|
78dcb0aa24 | ||
|
|
7fc66e8161 | ||
|
|
1678386f50 | ||
|
|
3859506f9b | ||
|
|
33b2605815 | ||
|
|
d0a887d98a | ||
|
|
24a92c8f3e | ||
|
|
f7852081b7 | ||
|
|
f0e24d8ce2 | ||
|
|
2724c922ce | ||
|
|
57406f6c14 | ||
|
|
8903a72662 | ||
|
|
ba7b8b7bef | ||
|
|
8c73128cd6 | ||
|
|
2abf328db8 | ||
|
|
fa8bb13a27 | ||
|
|
6f64e557e5 | ||
|
|
97a1238c85 | ||
|
|
aa0521602d | ||
|
|
2bc16d5fd9 | ||
|
|
d60a164025 | ||
|
|
727817410c | ||
|
|
28d3bd7da8 | ||
|
|
d0f9bdd251 | ||
|
|
ff98006471 | ||
|
|
5f1be90ae9 | ||
|
|
3da8bb7046 | ||
|
|
ac574d6c64 | ||
|
|
d9e6ebb20a | ||
|
|
88e5b7b200 | ||
|
|
73607663fd | ||
|
|
458ca578e7 | ||
|
|
884713cca5 | ||
|
|
a220c28a14 | ||
|
|
0761de3dae | ||
|
|
c2eaba990b | ||
|
|
e214691429 | ||
|
|
773b405429 | ||
|
|
2df9355367 | ||
|
|
cd64b0a3f7 | ||
|
|
2d573d8a34 | ||
|
|
51b4a1b758 | ||
|
|
4205f6930f | ||
|
|
12de3c5164 | ||
|
|
9af12afb57 | ||
|
|
fe3bd0074c | ||
|
|
3b55957d79 | ||
|
|
1a99b5836c | ||
|
|
035bfbc2fe | ||
|
|
4c705094f7 | ||
|
|
c376534a50 | ||
|
|
2c3ccdf030 | ||
|
|
475436242c | ||
|
|
613b774bf1 | ||
|
|
60c9af0599 | ||
|
|
2d842ded35 | ||
|
|
5fc391a47c | ||
|
|
a0298cf2b1 | ||
|
|
5bb489addb | ||
|
|
19ffe9b7a8 | ||
|
|
95dc6fe944 | ||
|
|
383f834704 | ||
|
|
e0d4477edc | ||
|
|
5cfb98fb8b | ||
|
|
3edf9aae2f | ||
|
|
55a80eab86 | ||
|
|
358aef16e3 | ||
|
|
1040f6c489 | ||
|
|
e271a65e79 | ||
|
|
3cdb4bf42e | ||
|
|
492f8d8ddf | ||
|
|
56209e7829 | ||
|
|
afb6754453 | ||
|
|
f9edb33d15 | ||
|
|
3730bc7df5 | ||
|
|
cfd771d1d8 | ||
|
|
75a028e825 | ||
|
|
9982a1325f | ||
|
|
1f32128ca9 | ||
|
|
e2034177c5 | ||
|
|
8520925e76 | ||
|
|
db9729e1fc | ||
|
|
2d3fc65758 | ||
|
|
5ddc028a2f | ||
|
|
470f75b08c | ||
|
|
211b872335 | ||
|
|
2c89359d42 | ||
|
|
962029bb3d | ||
|
|
b45a96358e | ||
|
|
29984c639d | ||
|
|
acb8d4b0aa | ||
|
|
7b947fa3f1 | ||
|
|
993710263d | ||
|
|
7bbe408e44 | ||
|
|
55dae31530 | ||
|
|
0af233c96c | ||
|
|
a7f74f374f | ||
|
|
0929694012 | ||
|
|
01b32ee6cd | ||
|
|
2936ba6e3d | ||
|
|
b1db5515d7 | ||
|
|
83033b4299 | ||
|
|
f865f74a0f | ||
|
|
fbee1b2d82 | ||
|
|
bcebc81fcd | ||
|
|
25f22b9839 | ||
|
|
97464bfa27 | ||
|
|
0e8b1981af | ||
|
|
5c25a52f95 | ||
|
|
409a6e65f9 | ||
|
|
9a9e542a7d | ||
|
|
5a9ff07f57 | ||
|
|
60e1bd52f7 | ||
|
|
fed6582d3e | ||
|
|
98d26e14d9 | ||
|
|
25fae9ad10 | ||
|
|
4a30f510e6 | ||
|
|
d0a5a583cd | ||
|
|
8dfc965d13 | ||
|
|
c9515b1d4c | ||
|
|
1380b023e2 | ||
|
|
e8f7772320 | ||
|
|
2f61be6e74 | ||
|
|
8b5a13435a | ||
|
|
3f0bfde54a | ||
|
|
0f3eea2fb5 | ||
|
|
a81f430e41 |
@@ -10,7 +10,7 @@
|
||||
"name": "codeman",
|
||||
"source": "./plugins/codeman",
|
||||
"description": "Drive Codeman from inside a Claude Code session: spawn worker sessions, prompt them, wait for them, read their answers, clean up. Acts only inside a Codeman-managed session.",
|
||||
"version": "1.30.0",
|
||||
"version": "1.33.2",
|
||||
"author": {
|
||||
"name": "Ark0N",
|
||||
"url": "https://github.com/Ark0N"
|
||||
|
||||
@@ -28,12 +28,15 @@ The frontend is plain JS served from `src/web/public/` with no bundler in dev: e
|
||||
CI runs all of these, so save yourself a round trip:
|
||||
|
||||
```bash
|
||||
npm run typecheck # tsc --noEmit, strict mode
|
||||
npm run typecheck # tsc --noEmit, strict mode
|
||||
npm run lint
|
||||
npm run format:check
|
||||
npm run check:frontend-syntax # syntax-checks the plain-JS frontend modules
|
||||
npm run check:frontend-syntax # syntax-checks the plain-JS frontend modules
|
||||
npm run check:browser-excludes # every browser-driven test is kept out of `npm test`
|
||||
```
|
||||
|
||||
`npm install` also installs a `pre-push` git hook that runs these static checks (about 10-40s, machine-dependent) and blocks the push if one fails. It skips itself when you push something other than the checked-out HEAD, or when the tree has uncommitted changes the checks would read. Skip it once with `CODEMAN_SKIP_PREPUSH=1 git push`; it never replaces a `pre-push` hook of your own.
|
||||
|
||||
### Tests
|
||||
|
||||
```bash
|
||||
|
||||
@@ -34,6 +34,13 @@ jobs:
|
||||
- name: Frontend JS syntax check
|
||||
run: npm run check:frontend-syntax
|
||||
|
||||
# Asks `vitest list` what CI would actually collect, rather than matching
|
||||
# filenames: a browser-driven test missing from BROWSER_TEST_GLOBS
|
||||
# (config/test-suites.ts) passes locally and dies in the test job with
|
||||
# "browserType.launch: Executable doesn't exist".
|
||||
- name: Browser-test exclusion check
|
||||
run: npm run check:browser-excludes
|
||||
|
||||
- name: Format check
|
||||
run: npm run format:check
|
||||
|
||||
@@ -100,6 +107,50 @@ jobs:
|
||||
fi
|
||||
echo "bash $BASH_VERSION: dsh identity probe survives a missing timeout"
|
||||
'
|
||||
# Installer v2: the question phase runs before the build, and every decision it
|
||||
# takes is bash logic over stubbed tailscale state. Drive the flags, the launch
|
||||
# default, the occupied-:443 menu and the rename question with canned answers,
|
||||
# so a bash-4 construct or a flipped default in any of them fails here, not on a
|
||||
# Mac. The JSON parsers need node (absent in this image) and are stubbed; their
|
||||
# own coverage is test/install-sh-invariants.test.ts plus the vitest gate.
|
||||
docker run --rm -v "$PWD":/w -w /w -e CODEMAN_INSTALL_SH_LIB=1 -e HOME=/tmp/h bash:3.2 bash -c '
|
||||
set -euo pipefail
|
||||
mkdir -p /tmp/h
|
||||
. /w/install.sh
|
||||
parse_flags --tailscale --service --name Build-Box --port 4000
|
||||
[[ "$CODEMAN_TAILSCALE" == "1" && "$LAUNCH_PRESET" == "2" && "$TS_NAME" == "Build-Box" && "$CODEMAN_PORT" == "4000" ]]
|
||||
[[ "$(ts_sanitize_name "$TS_NAME")" == "build-box" ]]
|
||||
has_tty() { return 0; }
|
||||
ANSWER=""; read_reply() { eval "$1=\"\$ANSWER\""; }
|
||||
systemctl() { return 0; }
|
||||
LAUNCH_PRESET=""; NONINTERACTIVE=0
|
||||
choose_launch_mode linux >/dev/null 2>&1
|
||||
[[ "$LAUNCH_CHOICE" == "2" ]]
|
||||
check_tailscale() { return 0; }
|
||||
ts_status_field() { case "$1" in "s.BackendState") printf Running ;; "s.Self && s.Self.DNSName") printf "box.tail.ts.net." ;; esac; }
|
||||
ts_backend_state() { printf Running; }
|
||||
ts_dns_name() { printf box.tail.ts.net; }
|
||||
ts_serve_443_target_port() { printf 8080; }
|
||||
ts_serve_find_port_mapping() { :; }
|
||||
ts_serve_port_used() { return 1; }
|
||||
detect_tailscale_serve_url() { :; }
|
||||
tailscale_choose_mapping >/dev/null 2>&1
|
||||
[[ "$TS_SERVE_MODE" == "path" && "$BIND_BASE_URL" == "/codeman" ]]
|
||||
RENAMED=""; tailscale_rename_node() { RENAMED="$1"; }
|
||||
TS_NAME=""; tailscale_choose_name >/dev/null 2>&1
|
||||
[[ -z "$RENAMED" ]]
|
||||
# A flag re-run keeps the password the unit already carries (and so
|
||||
# never writes the unauthenticated ack), and the hand-start line the
|
||||
# done screen prints carries every non-default value.
|
||||
read_existing_binding() { EXISTING_FOUND=1; EXISTING_HOST=0.0.0.0; EXISTING_PASSWORD=s3cret; EXISTING_ACK=0; EXISTING_BASE_URL=""; }
|
||||
CODEMAN_HOST=0.0.0.0; CODEMAN_TAILSCALE=0; unset CODEMAN_PASSWORD; BIND_ACK=0
|
||||
choose_network_binding >/dev/null 2>&1
|
||||
[[ "$BIND_PASSWORD" == "s3cret" && "$BIND_ACK" == "0" ]]
|
||||
BIND_HOST=0.0.0.0; BIND_PASSWORD=x; BIND_ACK=0; BIND_BASE_URL=/codeman; CODEMAN_PORT=4000
|
||||
[[ "$(start_command_hint)" == "CODEMAN_HOST=0.0.0.0 CODEMAN_PASSWORD="*" CODEMAN_BASE_URL=/codeman CODEMAN_PORT=4000 codeman web" ]]
|
||||
RECONFIGURE=0; parse_flags --port 4001; [[ "$RECONFIGURE" == "1" ]]
|
||||
echo "bash $BASH_VERSION: question phase (flags, launch default, occupied :443, rename opt-in, kept password, start line) ok"
|
||||
'
|
||||
|
||||
- name: CLI catalogue artifacts are in sync with stock.ts
|
||||
run: npm run generate:cli-catalog -- --check
|
||||
|
||||
+249
@@ -1,5 +1,254 @@
|
||||
# aicodeman
|
||||
|
||||
## 1.33.2
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- e439cf0: ### Thanks
|
||||
- @aakhter for keeping web-tab events private to their owner in multi-user mode (#501), with end-to-end isolation tests that fail without the fix, and for the browser-test exclusion check and pre-push hook (#500), including the hooks-dir resolution that never writes outside the repo's own `.git/hooks`.
|
||||
- @opticon454 for keeping CLIs installed from Settings across Docker container updates (#490) and for the static Git identity for the Docker images (#492).
|
||||
- @timkjr for letting a Shell pane's scroll-to-top reach tmux history (#494), with tests that fail on the commit before each fix.
|
||||
- @JDProfresh for tracking down why wheel and touch scrolling did nothing in Claude's default inline view (#498), with the tmux measurements that proved it.
|
||||
- @irisitymichaelgrundberg for the follow-up that makes an agent waiting on artifact comments raise its alert again (#491).
|
||||
|
||||
**A tab stays busy while Claude waits for its own workers.** When Claude hands work to an ultracode workflow or background agents, it ends its turn with `✻ Waiting for 1 dynamic workflow to finish` and resumes by itself when they report back. The idle probe used to call that session idle for the whole wait, and at phone width nothing on screen changes for minutes. A new optional registry field, `capabilities.workDetect.awaitingLine`, names that closing row, and only the newest column-0 row directly above the composer counts, so the session goes idle normally once the follow-up turn ends.
|
||||
|
||||
**Prompts sent through the API are no longer left unsent.** A prompt posted to `POST /api/sessions/:id/input` without `useMux` was written into the pane in one piece, and Claude Code (measured on 2.1.283) takes a burst of about a hundred characters or more as a paste, so the trailing `\r` became a newline and the prompt sat on the composer while the route answered 200. Short prompts went through, which is why it looked random; Codex and OpenCode showed the same thing. A plain prompt (printable text plus exactly one trailing `\r`) now goes through tmux: the text is typed, Enter is pressed as its own key, and the server presses it again while the prompt is still on the composer. Raw frames (escape sequences, a bracketed paste, a line feed, a bare `\r`) and an explicit `"useMux": false` keep the direct write. The same fix reaches cron jobs in "Paste (direct)" input mode, which reported `prompt_sent` for a prompt that never left the composer: the text is written raw, Enter follows as its own write 300 ms later, and the session presses it again while the prompt is still unsent. A cron run with no session to write to now fails instead of reporting the prompt as sent.
|
||||
|
||||
**Scrolling works again in Claude's default inline view (#498).** Wheel and touch gestures were forwarded to every Claude 2.1.187+ session as mouse reports, but only Claude's fullscreen renderer (`CLAUDE_CODE_NO_FLICKER=1`, or `"tui": "fullscreen"` in `~/.claude/settings.json`) listens for them, so in the default view scrolling did nothing. Codeman now forwards them only while Claude has mouse tracking switched on, and otherwise scrolls the terminal's own scrollback.
|
||||
|
||||
**Shell panes scroll back into tmux history (#494).** Scrolling to the top of a Shell pane now pulls the most recent 1 MiB of its tmux history, so output that arrived in a burst is reachable without pressing **Load full history**, which still loads the rest.
|
||||
|
||||
**An agent waiting on artifact comments alerts again (#491).** A session whose agent published an artifact and is waiting for somebody to comment on it now raises the normal idle alert and lands in NEEDS YOU, instead of being treated as busy with background work.
|
||||
|
||||
**Web-tab changes stay private in multi-user mode (#501).** The `webview:changed` event reached every connected user, exposing the ids of other users' web-tab creates, edits and deletes. It now carries the tab's owner and reaches that owner plus admins only. Single-user mode is unchanged apart from a new optional `owner` field on the event.
|
||||
|
||||
**Phone header tabs look like tabs (#504).** On phones every header tab is now a chip with a fill and a border, the Alt+N digit (a keyboard hint a phone cannot use) is hidden, names get 80px instead of 50px, and the strip fades at whichever edge still has tabs scrolled out of view.
|
||||
|
||||
**Docker: CLIs installed from Settings survive container updates (#490).** On the Compose deployment, CLIs installed from App Settings (DeepSeek, Pi and other npm-based CLIs) now go to `~/.local` on the persistent home mount, and `~/.local/bin` is on the image PATH, so recreating the container no longer discards them. Anything installed from Settings before this release has to be installed once more after the rebuild.
|
||||
|
||||
**Docker: a static Git identity for the server and agent images (#492).** Set `GIT_USER_NAME` and `GIT_USER_EMAIL` in `docker/.env` (or `CODEMAN_AGENT_IMAGE_GIT_USER_NAME` / `CODEMAN_AGENT_IMAGE_GIT_USER_EMAIL` on a bare-host install) and the identity is written to `/etc/gitconfig` in the server image and the Docker-case agent image. A half-set pair is refused on every build path. Both Docker changes edit `server.Dockerfile`, so the in-app updater asks Compose deployments to rebuild with `docker/Start-Codeman.sh` instead of updating in place.
|
||||
|
||||
**Contributor tooling (#500).** `npm run check:browser-excludes`, now a CI step, fails when a test that drives a real browser is still collected by `npm test`. `npm install` also installs a pre-push hook that runs the static CI checks before a push; it steps aside when the pushed ref is not HEAD or the tree has uncommitted changes the checks would read, and `CODEMAN_SKIP_PREPUSH=1 git push` skips it once.
|
||||
|
||||
**Fixes applied while landing.** A Shell pane's scroll-to-top (#494) no longer re-pulls the same window on every gesture once the browser's 50,000-row scrollback is full, and a pull that hit the byte cap no longer claims the older history is gone. The scroll-routing diagnostics (#498) now log whether Claude has mouse tracking on. The artifact-comment check (#491) also refuses a footer cut off in the middle of the chip. The pre-push hook (#500) steps aside when `npm` is not on PATH, as in some GUI git clients, instead of blocking every push. The Docker Git identity error (#492) names the two variables to set. New tests pin the image PATH order, the identity on both agent-image build paths, and the `?full=1&tail=` terminal route.
|
||||
|
||||
## 1.33.1
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- ### Thanks
|
||||
- @irisitymichaelgrundberg for closing sessions whose agent exited cleanly (#486), built carefully around every way a pane exit can lie (a SIGKILL with no status, a single misread), with the `.claude-images` guard split into its own commit as asked.
|
||||
- @opticon454 for the live-refreshing case picker and Manage search (#483), and for the uv/uvx, libsecret and pnpm additions to the Docker images (#487, #485).
|
||||
|
||||
**Finished sessions close themselves (#486).** A session whose agent you ended with `/exit` is now closed the same way the X button closes it, so finished sessions stop piling up on the board; the conversation stays resumable from the Resume list and the lifecycle log records "agent exited cleanly (status 0)". Only an explicit exit status 0 with no signal, confirmed by two pane reads, qualifies: a crashed or OOM-killed agent keeps its row with the exit code on the tab. The phone overview and desktop home rail now say `exited` instead of `idle`, reboot restore no longer offers to rebuild a session whose agent had exited, and closing one session no longer deletes the `.claude-images` directory that a sibling session in the same case still uses. Thanks @irisitymichaelgrundberg.
|
||||
|
||||
**Search in the phone Select Case sheet (#488).** The bottom sheet gains a "Search cases" field that filters by name (every word must match, any order, ignoring case), Enter picks the case when exactly one row is left, and Escape clears then closes. Also fixes a dead band under Create New Case and a list shorter than the sheet could show.
|
||||
|
||||
**An oversized paste no longer jams a session's input (#484).** A single input over the 64 KiB frame limit used to be refused by both transports, retried every 2 s forever, block every later input for that session and come back from localStorage on each reload. Pastes over the limit are now split into in-limit frames delivered in order (up to 1 MiB; larger ones are refused with a toast and never queued), a refused frame is dropped instead of retried, frames persisted by an older build are pruned on load, and the WebSocket answers an oversized frame with an explicit `too_large` error instead of silence.
|
||||
|
||||
- 8841bcc: Add a search box to the Manage tab of the Add Case dialog. It filters the case list by name or path, and the reorder arrows are disabled while a filter is active so a swap cannot involve a hidden case.
|
||||
- 8841bcc: The case picker now refreshes its list from `/api/cases` when it opens and every 5 seconds while it stays open, so folders deleted or created on disk appear without a page reload. If the selected case has been removed, the picker falls back to another case without saving it as the last-used one.
|
||||
|
||||
Thanks @opticon454.
|
||||
|
||||
- 77ba41f: Install `uv` and `uvx` in the Compose server image and the agent image, so MCP servers launched with `uvx` (such as the Nginx Proxy Manager MCP) can be enabled by Codex instead of failing with `uvx` not found. Both images also install `libsecret-1-0`, the native library the `keytar` dependency of the Azure DevOps MCP (`@azure-devops/mcp`) needs; without it the server crashes before answering the MCP initialize handshake.
|
||||
|
||||
The Compose server image now also carries `pnpm`: `dsh plugin` spawns a literal `pnpm` with no npm fallback, so the Run menu's "DeepSeek - add a terminal profile" button failed with `dsh: pnpm not found on PATH` there. Because this release changes `server.Dockerfile`, the in-app updater asks Compose deployments to rebuild the image (`Update-Codeman.sh`) rather than applying it in place.
|
||||
|
||||
Thanks @opticon454 (#487, #485).
|
||||
|
||||
## 1.33.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- CLI management from Settings (#476, finishing the CLI registry work from #343). `~/.codeman/clis.json` used to be hand-edit only; with the new opt-in `cliManagementEnabled` switch (synced, default OFF) App Settings → Agents & CLIs can enable or disable any CLI, install a missing stock CLI with its vetted install command, and add, edit or remove custom CLIs. Six new endpoints back it (`GET`/`POST /api/clis`, `PUT /api/clis/:id`, `POST /api/clis/:id/install`, `PUT /api/clis/custom/:id`, `DELETE /api/clis/:id`), documented in `docs/api-reference.md`. Every write is refused while the switch is off, is admin-only in multi-user mode, is serialized on one queue, and refuses to overwrite a `clis.json` that does not parse or has group/world permission bits. A custom entry is re-validated through the same schema as the stock ones and its install text is never executed. `shell` cannot be disabled. The Run menu and the welcome screen are now built from the enabled catalogue, so the welcome screen also offers Codex, Shell and any custom CLI, and the stock Claude entry is labelled "Claude Code".
|
||||
|
||||
Models: Opus 5.5 (`claude-opus-5-5`, 1M context capable) is offered in App Settings → Models and in task routing (#480).
|
||||
|
||||
Self-update: on a macOS `launchd-daemon` install, a Homebrew node upgrade could leave `update-status.json` stuck at `queued`, which made every later update fail with "An update is already in progress." The updater now falls back to `node` on PATH when the server's own node binary is gone, and an in-flight status that has not been written for 15 minutes is failed on the next read. A graceful shutdown that hangs is now force-exited after 10 s (and the launchd updater SIGKILLs a server that has not exited after 30 s), so launchd can start the new build instead of leaving the service down (#478). Both fixes protect updates that start FROM this release.
|
||||
|
||||
Session Manager (Cmd+K): rows keep their `mode`, `claudeSessionId` and `resumeId`, so the ⋯ menu's Resume session relaunches a Codex row as Codex on its own conversation, and the mode badge shows as it does on the home list (#477).
|
||||
|
||||
Maintainer fixes applied while landing #457: renaming a tab to the name it already has (the Session Options field saves on blur) is now a no-op, so it no longer pins the placeholder as the `/resume` title again; Docker sessions skip the transcript title sync, since their transcript lives in the container; and the agent skill's messaging examples no longer use a `w<N>-` name as the peer name.
|
||||
|
||||
Tests: the suite strips every inherited `CODEMAN_*` variable, so running it inside a Docker Compose deployment no longer writes into the deployment's real case root (#479).
|
||||
|
||||
### Thanks
|
||||
- @opticon454 for CLI management (#476), the last piece of the CLI registry, with every review item answered in one round, and for splitting the test isolation fix out into #479.
|
||||
- @shenlvkang-collab for the `/resume` title fix (#457) and the careful diagnosis behind it.
|
||||
- @julian3xl for the Session Manager row fix (#477), their first contribution.
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- 69a7128: fix(sessions): stop pinning the `w1-myapp` placeholder as Claude's session title. Local Claude spawns passed the tab name as `--name`, which is also the `/resume` picker entry and the terminal title, and a pinned title stops Claude generating its own, so every conversation of a case showed up in `/resume` as the same `w1-myapp` and none got a generated title. Only a name the user chose is pinned now; placeholder and auto-named tabs let Claude title the conversation again. Renaming a Claude tab also reaches `/resume`: the new name is appended to the conversation's transcript as the `custom-title` row `/rename` writes (a tab that was spawned with `--name` keeps re-appending its own title until its next respawn, so the rename wins from then on). Orchestrators that rely on a fixed peer name should give workers a descriptive `sessionName` rather than a `w<N>-` one.
|
||||
|
||||
## 1.32.1
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- 13e652e: Terminal copy: copying text out of a Claude Code or Codex pane no longer puts the pane's two-column transcript gutter on the clipboard, so pasted lines arrive flush instead of indented (#469). The width comes from the CLI registry (`capabilities.transcriptGutter`, 2 for claude and codex, measured on live panes) and is only a ceiling: a selection only ever shifts as a block, so its own indentation survives. Other CLIs and shells are untouched. It works in split panes and detached session windows too, and can be turned off per device in App Settings under Selection & clipboard.
|
||||
- 13e652e: Sessions: recovering a Claude session whose tmux pane had died relaunched `claude --session-id <id>`, which Claude refuses once that id has a transcript, so the pane died again straight away and the conversation was stranded. The relaunch now resumes the conversation (`--resume <id> || --session-id <id>`), including when tmux lost the whole session (#467).
|
||||
- 00f022c: Terminal: when a burst of output overflows the render queue and a frame has to be dropped, the repaint that repairs it is now retried until it actually happens, instead of being scheduled once and silently skipped when another load was in flight (#470).
|
||||
- 13e652e: Mobile: a long press on blank terminal space on Android Chrome no longer opens the keyboard and blanks the terminal (#471, fixes #360). The long-press guards are now armed before the press is checked for selectable text, so a press on empty space is swallowed the same way a press on a word already was.
|
||||
- 13e652e: Sessions: a tab whose agent has exited (the CLI quit, but tmux kept the pane) now says so with a muted dot and an `exited (137)` badge, instead of looking like an idle session (#466, part 1 of #446). The state is published as `paneExit` on the session and survives a restart. Nothing closes such sessions yet; that is part 2.
|
||||
- 13e652e: Docker: optional GitHub CLI and Azure CLI for private repositories (#472). Both are off by default. With `CODEMAN_INSTALL_GH=1` / `CODEMAN_INSTALL_AZ=1` as build args in `docker-compose.override.yml`, the server image gets `gh` and/or `az` (with the `azure-devops` extension) wired in as git credential helpers, so after one `gh auth login` or `az login` from a shell session, Add Case → Clone Repo can clone private GitHub and Azure DevOps repositories. `CODEMAN_AGENT_IMAGE_INSTALL_GH` / `_AZ` do the same for the Docker-case agent image, and only then are the sign-ins copied into new case containers. In multi-user mode a non-admin's clone runs with the credential helpers cleared. This changes `server.Dockerfile`, so Compose deployments need a `Start-Codeman.sh` rebuild rather than an in-app update.
|
||||
- 13e652e: Run menu: the Gemini, Antigravity and OMP run buttons now show their own colours on every skin; they rendered in Claude blue on all skins except OG (#463). The CLI registry's `accent` values were also corrected to the colours the UI really paints, and a test now guards the stylesheet trap that caused it.
|
||||
- 13e652e: Terminal: five ways the browser terminal could silently stop being correct are fixed (#431, #464). The browser terminal and the PTY can no longer disagree about their width, which is what produced doubled lines and half-overwritten text ("text gets muffled sometimes"): there is now one function that sizes the terminal, and every resize is answered with the geometry the PTY really holds. A replay clear goes through the terminal's own queue, so bytes written just before it no longer fuse into the next snapshot. A renderer that stops painting after an iOS PWA is backgrounded heals itself instead of needing a reload. Every terminal capture has a deadline that also covers the response body, and a capture that runs out of time during a tab switch falls back to the bounded tail instead of leaving a blank pane. Output lost to a half-open WebSocket is repainted on the next successful open. The service worker's precache list is now generated by the build and its cache is rotated per build, so old releases' assets no longer pile up.
|
||||
- 13e652e: Docker: new `docker/Update-Codeman.sh` for the major-update path the docs used to describe by hand (#465). It rebuilds the image with `--no-cache` before taking the stack down, clears the build-artefact volumes, refuses to run when another checkout's Compose project already owns the same name, and then hands over to `Start-Codeman.sh`.
|
||||
- 13e652e: Approvals: a session that is idle only because it is waiting on its own background work (Claude Code's `1 monitor` footer chip, or a Codex background terminal) no longer raises the yellow NEEDS YOU alert or a push (#473, fixes #468). Its idle item is opened already acknowledged, and the tab, the home screens and the rail show a small `watching` badge next to the state instead. The item still exists in the Approvals Inbox, and the TUI's pending count now leaves acknowledged items out.
|
||||
- b404dac: Maintainer fixes applied while landing this batch:
|
||||
- Terminal (#431): while another device holds the pane's width, a resize retry no longer re-fits xterm to the container and re-wraps the whole buffer every 30 s, and no longer clears scrollback for a redraw that never comes. The PTY's spawn geometry is now recorded at attach, so `ptyGeometry` never reports a size the PTY never held.
|
||||
- Terminal (#470): the `TERMINAL DROP` crash-trail line is logged once per recovery window instead of once per dropped frame (which wiped the rest of the trail within a second), and a refresh that died at its fetch deadline is no longer retried.
|
||||
- Sessions (#467): the resume pin also covers the branch where tmux lost the whole session, the conversation id Codeman reports follows what the relaunch actually resumed, and the test setup strips `CLAUDE_CONFIG_DIR` so the suite stays green for anyone running a separate Claude config dir.
|
||||
- Sessions (#466): detailed sidebar and rail rows show an `exited` pill instead of `idle`, the exit is announced to screen readers, and the user manual's tab-appearance table lists the new state.
|
||||
- Approvals (#473): a failed pane capture clears the `watching` badge rather than keeping a stale one (a failure now falls toward an alert, not toward silence), and the header bell's count leaves acknowledged items out, matching the TUI.
|
||||
- Run menu (#463): the Gemini and Antigravity run buttons no longer render two-tone on phones, Gemini's registry accent matches its tab badge, and a test now guards the stylesheet trap for every run mode.
|
||||
- Docker (#465): `Update-Codeman.sh` removes exactly the two build-artefact volumes it names instead of every named volume in the project, reports a failing `docker compose` instead of exiting silently, and its docs and comments were corrected. (#472): the multi-user notes say that a non-admin's seeded Docker case also receives the gh/az sign-in when those switches are on.
|
||||
|
||||
### Thanks
|
||||
- @irisitymichaelgrundberg for four PRs in this release: the `watching` badge that stops background work from raising false alerts (#473, from their own report #468), the exited-agent badge (#466) and the dead-pane resume fix (#467), both from their report #446, and the transcript-gutter strip for copied text (#469), a follow-up to their #451.
|
||||
- @rounakdatta for the terminal resilience work (#431) and the dropped-frame recovery (#470), both from their report #464, and for answering four rounds of review in full.
|
||||
- @opticon454 for private-repository support in the Docker images (#472), the `Update-Codeman.sh` script (#465) and the run-button colour fix (#463).
|
||||
- @DodgyBadger for the Android long-press fix (#471), from their own report #360.
|
||||
|
||||
## 1.32.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- d47f93a: feat(custom-model): the model picker puts the ready model first
|
||||
|
||||
When a custom endpoint has more than one model, the Run menu's picker now promotes one row to the top instead of showing raw discovery order: the model llama-swap reports loaded and ready right now (tagged "Currently loaded", the one a launch attaches to with zero wait), else the model you last launched on that harness and endpoint (tagged "Last used", remembered per device). The endpoint's default keeps its own pill, nothing is ever auto-chosen, and a plain OpenAI-compatible server or an endpoint that does not answer within a second simply keeps the old order. The probe is bounded on the client too, so a GPU box that is off no longer holds the picker closed for five seconds.
|
||||
|
||||
- d47f93a: feat(split-pane): view two live sessions side by side
|
||||
|
||||
A new Split button in the header (opt-in in App Settings, off by default, desktop only at 1180px and wider) opens a picker and shows a second live session beside the active one: its own terminal, its own WebSocket, and a divider you can drag. When either session ends the view collapses back to one pane, with Pane B promoted to the primary when it is Pane A that ended. Nothing is persisted on purpose in this first cut, so a page reload always returns to a single pane. Pane B is deliberately plainer than the primary pane (no local-echo overlay, CJK input, touch handling or keyboard accessory bar); the design and the v2 boundaries are in discussion #452.
|
||||
|
||||
- 72d437a: Installer v2. `curl -fsSL https://getcodeman.com/install | bash` now looks at the machine first, asks at most three questions up front (how the dashboard is reached, optionally what to call the machine on your tailnet, whether to run Codeman as a background service), does the install unattended behind progress spinners with the output in `~/.codeman/install.log`, and ends on the URL with a QR code to scan. One consent covers every missing package and sudo asks for your password once. Flags pipe through `bash -s --` (`--tailscale | --lan | --local`, `--name <n> | --no-rename`, `--service | --run | --no-start`, `--yes`, `--password`, `--port`), `install.sh status` prints the URL and the QR code again, and the cloudflared question moved out of the main flow into `install.sh cloudflared`. On the Tailscale route, a `:443` that already belongs to another app gets Codeman under `https://<node>/codeman` (or on a second port) instead of a dead end, the node can be renamed opt-in (`--name`, `install.sh name`, undone by uninstall), and the HTTPS-certificates toggle is polled with the admin page opened for you. Also fixed on the way: the installer's own `npm install` no longer lets the postinstall start a stray server on port 3000 (the service crash-looped on EADDRINUSE while the done screen said "running"), the LAN address comes from the default route rather than the first interface, a hand-written LaunchDaemon on a headless Mac is left alone, a flag re-run keeps an existing dashboard password, and the done screen's start command carries the sub-path and port it was installed with.
|
||||
- d47f93a: feat(mobile): a Compose key for writing prompts on a phone
|
||||
|
||||
The agent keyboard bars on phones replace their Paste key with Compose: a real multiline editor with autocorrect and spellcheck, per-session drafts kept in memory only, image attach that never writes into the terminal early, and a Send that delivers the text as one paste followed by Enter, so a long prompt no longer has to be typed blind into the terminal composer. Anything you had already typed into the terminal is picked up into the editor. Shell sessions keep the direct Paste key. This is the manual first slice from #359; the auto-open setting and terminal tap routing are a separate follow-up.
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- d47f93a: refactor(run-menu): one table-driven launcher for every external CLI
|
||||
|
||||
The eight near-identical per-CLI launch functions in the Run menu collapsed into one launcher driven by a table that a CI test keeps in step with the CLI registry, and a second no-id-branching guard now covers the frontend the way the backend guard covers the server. No behaviour change: the refactor was verified byte-identical across 288 launch permutations against the previous code.
|
||||
|
||||
- e899af4: Maintainer fixes applied while landing the above. The model picker's promoted row keeps its Default pill (the promotion tag and the default marker are two pills now, and they render as pills in the picker rather than as plain text). The phone composer keeps its bottom gutter on folding devices (the generic fold rule used to erase it), a whitespace-only draft is no longer sent, and its dialog is translated on a zh-CN UI. A split that collapses mid-drag no longer leaves the page stuck in resize-cursor mode, Pane B refuses a session that has no live process, and a burst of refresh frames replays once instead of twice. The `</head>` script injections on the page render use replacer functions, so a CLI label containing `$'` can no longer splice the document into the inline script, and the frontend no-id-branching guard now catches comparisons on any variable name.
|
||||
- 6ef71ec: ### Thanks
|
||||
- @timkjr for split-pane sessions (#453): five review rounds turned around in two days, and the pointer-capture edge case measured in a real browser rather than reasoned about.
|
||||
- @DodgyBadger for the mobile prompt composer (#444), a first contribution that took the scope back down to one slice when asked, and that verified the delivery path against a live tmux pane and a live Claude Code composer instead of trusting the diff.
|
||||
- @opticon454 for putting the ready model first in the picker (#459) and for collapsing the eight Run-menu launch functions into one (#458), proven byte-identical across 288 launch permutations instead of argued.
|
||||
|
||||
## 1.31.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- 035bfbc: feat(remote): wake a sleeping remote host from Codeman
|
||||
|
||||
A remote SSH case pointing at a machine that suspends used to fail the same way every
|
||||
time: the session was there, the host was not, and typing into it went nowhere. A host
|
||||
can now carry a wake target, either a MAC address for Wake-on-LAN (Codeman builds the
|
||||
magic packet itself, so nothing reaches a shell) or a wake command of your own, and
|
||||
Codeman uses it when you ask for the host: when you type into a sleeping session, when
|
||||
you press the wake button on the banner, or when you start or attach a session on that
|
||||
host. Input you type while it wakes is buffered and flushed once it is back, up to 4 KB,
|
||||
and a chunk over that is refused outright rather than delivered as a fragment.
|
||||
|
||||
Waking only ever happens because you asked. No watcher, dropped-session handler or
|
||||
boot-recovery path can reach it, since a machine woken by a reconnect watcher would come
|
||||
back seconds after every suspend.
|
||||
|
||||
- fbee1b2: feat(custom-model): pick a custom endpoint straight from the Run menu
|
||||
|
||||
#393 landed the backend for custom model endpoints and left it reachable only over the
|
||||
HTTP API. This is the rest of it. Turn on Custom model endpoints in App Settings, save
|
||||
an endpoint, and the Run dropdown grows a Custom Endpoints section built live off the
|
||||
CLI registry, one entry per harness that can actually redirect plus each endpoint you
|
||||
saved. Pick one and it launches that harness pointed at your server, asking which model
|
||||
first when the endpoint has more than one. Endpoints re-discover themselves every five
|
||||
minutes, and one unreachable endpoint never blocks the others. App Settings gains full
|
||||
add, edit and delete for endpoints.
|
||||
|
||||
Seven of the harnesses (opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP) now launch
|
||||
directly onto the endpoint with no restart at all, where before you watched a native
|
||||
boot followed immediately by a second one. Claude still launches and then restarts in
|
||||
place, which its own resume makes far less jarring.
|
||||
|
||||
Most of this release's work went into things that only show up against a real server,
|
||||
and each was found that way rather than in tests: a freshly launched CLI reporting
|
||||
itself busy for its own startup and getting refused; Claude Code assuming a large
|
||||
context window for a model it does not recognise and silently overflowing a small one;
|
||||
a model whose real context is below what Claude Code's own system prompt costs, which
|
||||
no setting can fix and which now warns before launching into a certain failure; and the
|
||||
big one, llama.cpp running exactly one model at a time, so applying a selection can
|
||||
unload the model another session is using. That last case now asks first, tells you
|
||||
which session it affects, and keeps a "loading model" notice on screen for the whole
|
||||
swap window, so a prompt sent mid-swap reads as loading rather than as an answer from
|
||||
whatever was loaded a moment ago. A background sweep also catches the reverse: your
|
||||
session's model being evicted later by somebody else's ordinary use.
|
||||
|
||||
Two things worth knowing if you drive this over the HTTP API or run multi-user. The two
|
||||
questions an apply can ask (the model's context window is too small, and loading it will
|
||||
unload the model another session is using) are now answered by separate
|
||||
`confirmedContext` and `confirmedSwap` fields rather than one `confirmed`. They shared a
|
||||
flag until now, and since the context check runs first, confirming that one silently
|
||||
agreed to evict another session's model as well. The old `confirmed` still means both.
|
||||
And `CLAUDE_CONFIG_DIR` is now admin-only in multi-user mode: it joined claude's
|
||||
privileged env keys, so a non-granted owner can no longer set it through `envOverrides`,
|
||||
and an already-persisted one is dropped on reboot-restore, which returns that session to
|
||||
the default Claude account rather than the per-client one it was pointed at. Single-user
|
||||
installs are unaffected.
|
||||
|
||||
Remote SSH and Docker sessions are refused for now, since their restart reattaches a
|
||||
durable tmux rather than relaunching the agent.
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- c9515b1: fix(terminal): keep the output a pane capture could not contain. Opening a session, a backpressure refresh, a clear-terminal reload and a full-history re-pull all load the screen from a tmux pane capture, and anything the CLI printed between that capture and the end of the load used to be dropped, so its next partial redraw landed on a frame the terminal had never seen: missing or garbled output right after a tab switch or a refresh, plainest in a shell session. Each load now replays exactly the output that arrived after the capture, through one shared rule for all four paths, and a refresh that restores your scroll position no longer snaps back to the bottom afterwards.
|
||||
- 3edf9aa: fix(terminal): replay a pane capture at the geometry it was taken at
|
||||
|
||||
Opening a session could draw a frame built for a pane bigger than your terminal. A
|
||||
taller pane wrote its overflow rows onto the last line and lost the rows underneath
|
||||
(against a 50-row pane, a 30-row terminal rendered 28 of a 45-line command and drew
|
||||
the survivors twice), and a wider one wrapped every row and scrolled the whole frame
|
||||
up by one. The terminal response now reports the geometry the capture was really
|
||||
taken at, so the browser can see the mismatch and replay once at the size that stuck.
|
||||
A pane that cannot be sized to fit is diagnosed once per session instead of on every
|
||||
tab switch.
|
||||
|
||||
- 035bfbc: ### Thanks
|
||||
- @irisitymichaelgrundberg for three terminal fixes in one release: keeping the output a pane capture could not contain (#436), replaying a capture at the geometry it was taken at (#435, five rounds and a Playwright suite that fails against the merge base), and trimming the padding out of a copied selection (#451), where the scan-instead-of-regex call avoided a 2.9s freeze nobody would have traced back to a copy.
|
||||
- @timkjr for a first contribution that found a real silent failure: the Instance count stepper next to the Run button had only ever applied to Claude, so on the other eight run modes it launched one session and said nothing (#454).
|
||||
- @Randalix for Wake-on-LAN on remote hosts (#439), built and live-tested against a real sleeping machine, and for reading the whole diff again between rounds rather than only the parts that were asked about.
|
||||
- @opticon454 for turning #393's backend-only custom model endpoints into the whole feature (#430), and for validating it against a real llama-swap box rather than against the tests: the `/props` versus `/running` context discrepancy and the DeepSeek `/v1` root cause were both tracked down to the SDK source instead of guessed at.
|
||||
|
||||
- c376534: fix(run): make the Instance count stepper work for every non-Claude mode
|
||||
|
||||
The Instance count stepper next to the Run button only ever applied to Claude.
|
||||
Setting it to 3 and launching OpenCode, Codex, Gemini, Antigravity, Pi, OMP, Grok or
|
||||
DeepSeek started exactly one session, with no error and no hint that the control had
|
||||
done nothing. All eight now launch the count you asked for, and the opening banner
|
||||
says how many are starting. The one exception is a launch started from the Custom
|
||||
Endpoints section of the Run menu, which always starts a single session.
|
||||
|
||||
- 19ffe9b: fix(input): make sure a prompt sent through the API actually leaves the composer. Claude Code 2.1.277 started ignoring Enter for the first 30 to 50 seconds after the composer paints while still accepting the typed text, so a prompt sent right after a session came up sat unsent in the pane and every waiter (send-and-wait, the agent skill, cron, the maintainer bot) burned its whole timeout on a turn that never started. The server now reads the pane after every programmatic write that carried Enter and presses Enter again, on a 2 to 60 second schedule, only while the composer verifiably still holds the text it sent; an empty composer, other text, or a pane with no composer at all ends it. The agent skill's `sendwait` gets the same loop for servers that predate this, and its preamble version moves to 1.30.1 so an already-seeded agent picks up the fresh copy.
|
||||
- f9edb33: fix(terminal): trim the padding out of a copied selection
|
||||
|
||||
Copying out of a pane put a wall of spaces on the clipboard. xterm hands back
|
||||
whole screen rows and trims only the cells that were never written to, so the
|
||||
real spaces a full-screen program paints across the unused part of a row count
|
||||
as content: measured against Claude Code in a 282-column pane, single lines
|
||||
arrived carrying 138 trailing spaces. Pasting that into a chat client or an
|
||||
editor meant deleting the whitespace by hand, while Windows Terminal, iTerm2 and
|
||||
GNOME Terminal all trim it for you. A copy now drops the trailing run from every
|
||||
line, on all four paths (the Ctrl+C chord, right-click, the phone selection
|
||||
button and Auto Copy), while leading indentation is left exactly as it is. An
|
||||
Alt+drag rectangular selection is copied verbatim, because its columns lining up
|
||||
is the point of that gesture. A selection holding nothing but padding is refused
|
||||
rather than copied as bare line breaks.
|
||||
|
||||
## 1.30.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
@@ -19,6 +19,10 @@
|
||||
<a href="https://github.com/Ark0N/Codeman/commits/master"><img src="https://img.shields.io/github/commit-activity/t/Ark0N/Codeman?style=flat-square&color=1e3a5f" alt="Total commits"></a>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
⭐ <strong>Like Codeman? <a href="https://github.com/Ark0N/Codeman">Give it a star on GitHub!</a></strong> It takes one click and helps more people find the project. ⭐
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<strong>English</strong> • <a href="README.zh-CN.md">简体中文</a>
|
||||
</p>
|
||||
@@ -61,12 +65,13 @@ The installer asks before every system change, and re-running the same line upda
|
||||
curl -fsSL https://getcodeman.com/install | bash
|
||||
```
|
||||
|
||||
This installs Node.js, tmux and a build toolchain if missing (node-pty ships no Linux prebuilds, so it compiles from source), clones Codeman to `~/.codeman/app`, and builds it. A few things worth knowing:
|
||||
This installs Node.js, tmux and a build toolchain if missing (node-pty ships no Linux prebuilds, so it compiles from source), clones Codeman to `~/.codeman/app`, and builds it. It looks at what is already on the machine, asks at most three questions, then does all the work unattended and ends on the URL with a QR code for your phone. A few things worth knowing:
|
||||
|
||||
- **It asks first.** Every system change (package installs, AI CLI download) is prompted, and a menu at the end lets you choose: run Codeman in this terminal, install it as a background service (systemd/launchd, auto-start on boot), or don't start yet. Nothing runs in the background unless you pick it.
|
||||
- **How it's reachable, your choice.** The installer offers three ways to reach the dashboard: **Tailscale** (loopback bind fronted by `tailscale serve`, so you get `https://<machine>.<tailnet>.ts.net` with a real certificate and your tailnet as the login, no password needed), **any device on your network** (`0.0.0.0`, with a strongly recommended password prompt), or **this machine only** (`127.0.0.1`, safest). Skipping the password on a network bind requires an explicit confirmation and ends with a loud warning. The highlighted default reflects what is already on the machine (Tailscale when it is already in use, your existing binding on a re-run), and a bare Enter never pulls in new software. A bare `codeman web` started by hand still defaults to loopback.
|
||||
- **Re-run to update.** The same one-liner updates a finished install in place: local changes in `~/.codeman/app` are stashed (never discarded), and a running service is restarted and verified. If a first install was interrupted, re-running resumes the full setup instead. `install.sh update` and `install.sh uninstall` also exist.
|
||||
- **CI / headless:** without a terminal attached, steps that would change your system abort with instructions instead of running silently. Set `CODEMAN_NONINTERACTIVE=1` to approve them for automation.
|
||||
- **Three questions, all up front.** How the dashboard is reached, optionally what to call this machine on your tailnet, and whether to run Codeman as a background service (systemd/launchd, auto-start on boot; Enter says yes). Everything that needs you, including one consent for all missing packages, one sudo password, and the Tailscale login, happens before the build, so you can walk away while it compiles.
|
||||
- **How it's reachable, your choice.** **Tailscale** (loopback bind fronted by `tailscale serve`, so you get `https://<machine>.<tailnet>.ts.net` with a real certificate and your tailnet as the login, no password needed), **any device on your network** (`0.0.0.0`, with a strongly recommended password prompt), or **this machine only** (`127.0.0.1`, safest). Skipping the password on a network bind requires an explicit confirmation and ends with a loud warning. The highlighted default reflects what is already on the machine (Tailscale when it is already connected, your existing binding on a re-run), and a bare Enter never pulls in new software. If another app already owns `:443` on your node, Codeman goes under `https://<machine>.<tailnet>.ts.net/codeman` or on a second port instead of replacing it. A bare `codeman web` started by hand still defaults to loopback.
|
||||
- **The name is yours to choose.** By default the URL uses the machine's existing tailnet name. Answering yes to the second question renames the machine to `codeman-<hostname>` (which also renames it for SSH, so the default is no); `install.sh name` does it later.
|
||||
- **Re-run to update.** The same one-liner updates a finished install in place: local changes in `~/.codeman/app` are stashed (never discarded), and a running service is restarted and verified. If a first install was interrupted, re-running resumes the full setup instead. `install.sh status` prints the URLs and the QR code again; `install.sh update`, `install.sh tailscale` and `install.sh uninstall` also exist.
|
||||
- **Flags for the impatient.** `curl -fsSL https://getcodeman.com/install | bash -s -- --tailscale --service` answers the questions from the command line (`--lan`, `--local`, `--run`, `--no-start`, `--name <n>`, `--port <n>`, `--yes` too). **CI / headless:** without a terminal attached, steps that would change your system abort with instructions instead of running silently; set `CODEMAN_NONINTERACTIVE=1` to approve them for automation.
|
||||
|
||||
You'll need at least one AI coding CLI installed — [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), [Gemini CLI](https://github.com/google-gemini/gemini-cli), [Pi](https://pi.dev), [Grok Build](https://github.com/xai-org/grok-build), [DeepSeek Harness](https://github.com/deepseek-ai/deepseek-harness), or [OMP](https://github.com/can1357/oh-my-pi) (any combination works; Gemini CLI is enterprise-only since Google's consumer cutover, and Antigravity is its successor). The installer detects whichever of the nine is present; if none is found, it offers to install any of them from a menu (DeepSeek excepted, since its npm package installs only a launcher with no runnable profile), or you can skip and install one yourself later. After install:
|
||||
|
||||
@@ -219,7 +224,7 @@ codeman web --https
|
||||
# Open on your phone: https://<your-ip>:3000
|
||||
```
|
||||
|
||||
> `localhost` works over plain HTTP. Use `--https` when accessing from another device, or use [Tailscale](https://tailscale.com/) (recommended): the installer can set it up for you (choose **Tailscale** at the network-access prompt, or run `bash ~/.codeman/app/install.sh tailscale` on an existing install). That gives you `https://<your-machine>.<tailnet>.ts.net` with a real certificate: private to your tailnet, no password required, and PWA install + push notifications work on your phone.
|
||||
> `localhost` works over plain HTTP. Use `--https` when accessing from another device, or use [Tailscale](https://tailscale.com/) (recommended): the installer can set it up for you (choose **Tailscale** at the network-access prompt, or run `bash ~/.codeman/app/install.sh tailscale` on an existing install). That gives you `https://<your-machine>.<tailnet>.ts.net` with a real certificate: private to your tailnet, no password required, and PWA install + push notifications work on your phone. The installer ends on that URL with a QR code to scan, and `bash ~/.codeman/app/install.sh status` prints it again any time.
|
||||
|
||||
### Secure QR Code Authentication
|
||||
|
||||
|
||||
@@ -23,6 +23,10 @@
|
||||
<a href="https://github.com/Ark0N/Codeman/commits/master"><img src="https://img.shields.io/github/commit-activity/t/Ark0N/Codeman?style=flat-square&color=1e3a5f" alt="Total commits"></a>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
⭐ <strong>喜欢 Codeman?<a href="https://github.com/Ark0N/Codeman">在 GitHub 上给它点个 Star 吧!</a></strong>只需轻点一下,就能帮助更多人发现这个项目。⭐
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/subagent-demo-20260724.gif" alt="Codeman — 并行子智能体可视化" width="900">
|
||||
</p>
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
[
|
||||
{
|
||||
"id": "claude",
|
||||
"label": "Claude",
|
||||
"label": "Claude Code",
|
||||
"shortBadge": "CC",
|
||||
"enabled": true,
|
||||
"order": 0,
|
||||
|
||||
@@ -27,7 +27,12 @@ export const BROWSER_TEST_GLOBS = [
|
||||
'test/webgl-fallback.test.ts',
|
||||
'test/terminal-copy-shortcut.test.ts',
|
||||
'test/terminal-keycode229-recovery.browser.test.ts',
|
||||
'test/capture-load-window.browser.test.ts',
|
||||
'test/capture-geometry-retry.browser.test.ts',
|
||||
'test/codex-predictive-echo.test.ts', // also needs a real codex binary
|
||||
'test/split-pane-terminal.browser.test.ts',
|
||||
'test/split-pane-orchestration.browser.test.ts',
|
||||
'test/split-pane-auto-collapse.browser.test.ts',
|
||||
];
|
||||
|
||||
/**
|
||||
|
||||
@@ -15,6 +15,12 @@ TZ=Australia/Perth
|
||||
# this value rebuilds the image with a matching account.
|
||||
CODEMAN_RUNTIME_USER=codeman
|
||||
|
||||
# Optional Git identity for commits made by Codeman and Docker-case agents. These values
|
||||
# are written to each image's system Git configuration when it is rebuilt, so
|
||||
# deployments can configure a consistent default. Set both values together.
|
||||
# GIT_USER_NAME=
|
||||
# GIT_USER_EMAIL=
|
||||
|
||||
# Required. Persistent Codeman application data, CLI credentials, and session
|
||||
# state are stored here on the host and mounted at the runtime account's home
|
||||
# directory in the container.
|
||||
@@ -50,6 +56,14 @@ CODEMAN_USERNAME=admin
|
||||
# README.md, "Reverse-proxy host allowlist".
|
||||
# CODEMAN_ALLOWED_HOSTS=codeman.example.com,.internal.example.com
|
||||
|
||||
# The GitHub CLI (gh) and the Azure CLI (az, with the azure-devops extension)
|
||||
# can be built into the images as git credential helpers, so Codeman can clone
|
||||
# private GitHub and Azure DevOps repositories. Both are OFF by default and are
|
||||
# NOT set here: turn them on in docker-compose.override.yml with the build args
|
||||
# CODEMAN_INSTALL_GH / CODEMAN_INSTALL_AZ and, for the Docker-case agent image,
|
||||
# the environment variables CODEMAN_AGENT_IMAGE_INSTALL_GH / _AZ. See
|
||||
# README.md, "Private repositories".
|
||||
|
||||
# Optional: authenticate Gemini CLI without an interactive login.
|
||||
GEMINI_API_KEY=
|
||||
|
||||
|
||||
@@ -41,6 +41,115 @@ Releases that change `server.Dockerfile`, `docker-compose.yaml`, or add a key to
|
||||
changed, and asks you to run `Start-Codeman.sh` here on the host instead. Details:
|
||||
[`../docs/docker-self-update.md`](../docs/docker-self-update.md).
|
||||
|
||||
### Major updates
|
||||
|
||||
`Start-Codeman.sh` rebuilds the image on every start, but with the layer cache,
|
||||
and it refreshes the build-artefact volumes selectively: `codeman-dist` when
|
||||
the checkout's HEAD moved, `codeman-node-modules` only when `package-lock.json`
|
||||
changed. That is right for an ordinary `git pull`. It is not enough when a
|
||||
`server.Dockerfile` change bumps the Node base image without touching the
|
||||
lockfile: `node-pty` is compiled from source (there is no Linux prebuild), so
|
||||
the old `codeman-node-modules` volume would keep a build made for the previous
|
||||
Node version. For that case, or whenever you want to be certain of what ships,
|
||||
`docker/Update-Codeman.sh` force-rebuilds the image with no layer cache, stops
|
||||
the stack, removes the `codeman-node-modules` and `codeman-dist` volumes, then
|
||||
hands off to `Start-Codeman.sh` for the usual start:
|
||||
|
||||
```sh
|
||||
bash docker/Update-Codeman.sh
|
||||
```
|
||||
|
||||
Pass `--keep-volumes` to skip clearing them (safe only if you know the
|
||||
rebuilt image's `node_modules`/`dist` did not change). The scripted default
|
||||
is the "Resetting the build artefacts" procedure in
|
||||
[`../docs/docker-self-update.md`](../docs/docker-self-update.md). Only those
|
||||
two volumes are removed, by name within this Compose project; any volume a
|
||||
`docker-compose.override.yml` adds is left alone, and application data and
|
||||
case workspaces are host bind mounts, never touched either way.
|
||||
|
||||
## Git commit identity
|
||||
|
||||
Set `GIT_USER_NAME` and `GIT_USER_EMAIL` in `docker/.env` before rebuilding:
|
||||
|
||||
```sh
|
||||
GIT_USER_NAME='Your Name'
|
||||
GIT_USER_EMAIL='you@example.com'
|
||||
```
|
||||
|
||||
Compose passes the values to the Codeman server build, and to the server process
|
||||
when it builds Docker-case agent images. Both images write the pair to Git's
|
||||
system configuration during their build, so commits retain the same identity
|
||||
after a container or agent image is recreated. Set both values together; an
|
||||
image build with only one value fails rather than using a partial identity. An
|
||||
identity already present in `CODEMAN_APPDATA_PATH`'s `~/.gitconfig` overrides
|
||||
the server image's system-level default.
|
||||
|
||||
Run `bash docker/Start-Codeman.sh` after changing the server values. Rebuild an
|
||||
existing agent image with `node scripts/build-agent-image.mjs --no-cache` in the
|
||||
server container, then recreate any Docker cases that should use it.
|
||||
|
||||
## Private repositories (GitHub and Azure DevOps)
|
||||
|
||||
The images can include the GitHub CLI (`gh`) and the Azure CLI (`az`, with the `azure-devops` extension), wired into the system Git configuration as credential helpers, so Codeman can clone private repositories. Both are **opt-in and off by default**, and are turned on per host in `docker-compose.override.yml`.
|
||||
|
||||
### Turning them on
|
||||
|
||||
Add the build arguments to `docker-compose.override.yml` (see [Local customisation](#local-customisation)), then rebuild with `Start-Codeman.sh`. Set only the one you need:
|
||||
|
||||
```yaml
|
||||
services:
|
||||
codeman:
|
||||
build:
|
||||
args:
|
||||
CODEMAN_INSTALL_GH: '1'
|
||||
CODEMAN_INSTALL_AZ: '1'
|
||||
environment:
|
||||
# The same two switches for the Docker-case agent image Codeman builds.
|
||||
CODEMAN_AGENT_IMAGE_INSTALL_GH: '1'
|
||||
CODEMAN_AGENT_IMAGE_INSTALL_AZ: '1'
|
||||
```
|
||||
|
||||
The `build: args:` pair controls the Codeman server image. The `environment:` pair controls the agent image for [Docker cases](../docs/docker-cases.md), which Codeman builds on the first Docker case; an agent image that already exists is not rebuilt by this, so run `node scripts/build-agent-image.mjs --no-cache` inside the container afterwards. The same variables work in front of that command when building it by hand. Values must be `0` or `1`; anything else stops the build with an error naming the argument.
|
||||
|
||||
They are not `.env` settings: turning a CLI on is a per-host choice, which is what the override file is for, and a new `.env.example` key makes the in-app updater refuse to update every existing installation until its `.env` gains the key.
|
||||
|
||||
The Azure CLI is the large one, about 600 MB of the roughly 670 MB the pair adds. A CLI left off leaves nothing functional behind: no apt repository, no package, no `azure-devops` extension and no credential-helper entry, so git for that host behaves exactly as it does without this feature. With both off the image is functionally unchanged; it still carries the `AZURE_EXTENSION_DIR` variable, an empty extensions directory and one small layer that copies and then removes the helper script.
|
||||
|
||||
### Signing in
|
||||
|
||||
With a CLI on, the system Git configuration routes credentials through it:
|
||||
|
||||
| Host | Credential helper | Sign in with |
|
||||
| ----------------------------------------------------- | ----------------------------------------- | ---------------------------- |
|
||||
| `https://github.com`, `https://gist.github.com` | `gh auth git-credential` | `gh auth login` |
|
||||
| `https://dev.azure.com`, `https://*.visualstudio.com` | `/usr/local/bin/git-credential-azure-cli` | `az login --use-device-code` |
|
||||
|
||||
Codeman itself still collects no Git credentials. Sign the container in once from a **Terminal / Shell** session (Run menu). The session runs as the runtime account, so the sign-in is stored under `CODEMAN_APPDATA_PATH` (`~/.config/gh`, `~/.azure`) and survives rebuilds and container recreation:
|
||||
|
||||
```sh
|
||||
gh auth login # GitHub.com -> HTTPS -> "Login with a web browser" (device code)
|
||||
az login --use-device-code # then: az devops configure --defaults organization=https://dev.azure.com/<org>
|
||||
```
|
||||
|
||||
After that, **Add Case → Clone Repo** accepts private `https://` URLs on those hosts, and `git clone` works from any session. Until a CLI is signed in its helper prints nothing, so a private clone fails immediately with the usual authentication error rather than waiting on a prompt.
|
||||
|
||||
**Multi-user mode:** every Codeman user's git runs as the same server account, so these sign-ins would otherwise be shared. Clone Repo therefore runs a **non-admin**'s clone and preflight with every git credential helper cleared (`git -c credential.helper=`): a non-admin can clone public repositories and anything their own SSH setup allows, but not a private https repository through the admin's `gh`/`az` sign-in. Admins, and single-user mode, keep the helpers. A non-admin's own agent sessions still run as that same account, and with the agent-image `gh`/`az` switches on, a non-admin's Docker case with credential seeding on also receives the server account's `gh`/`az` sign-in, the same as the Claude and Codex credentials; see `docs/security-architecture.md`, multi-user mode.
|
||||
|
||||
Azure DevOps is authenticated with an Entra ID access token that the helper requests from `az` for each Git operation, so nothing is written to disk beyond `az`'s own sign-in. An account that has to use a personal access token can set `AZURE_DEVOPS_EXT_PAT` for the container instead (for example under `environment:` in `docker-compose.override.yml`); the helper prefers it when present. SSH remotes are unaffected by any of this and keep using the account's own keys.
|
||||
|
||||
Docker cases copy these sign-ins into a case container only when the matching agent-image switch is on (`CODEMAN_AGENT_IMAGE_INSTALL_GH=1` for `~/.config/gh/hosts.yml` and `config.yml`, `CODEMAN_AGENT_IMAGE_INSTALL_AZ=1` for the sign-in files from `~/.azure`) and the case has credential seeding on. With a switch off they are never copied, even when the files exist, because a GitHub token or an Azure refresh token is usable by anything in the container. The copies are made when the container is **created**, so an existing case container never picks them up: after turning a switch on, signing in, or rebuilding the agent image, **recreate the case container** (remove it; the next session in that case creates a fresh one).
|
||||
|
||||
The GitHub agent skill for `gh` installs into the runtime account's home in the same session:
|
||||
|
||||
```sh
|
||||
gh skill install cli/cli gh --scope user
|
||||
gh skill update gh # after a later gh release
|
||||
```
|
||||
|
||||
### Versions
|
||||
|
||||
Both CLIs, and the extension, are installed from their vendors' repositories with no version pinned, so they arrive at whatever is current when that build step runs. Docker caches the step, though: `Start-Codeman.sh` rebuilds with the cache, which keeps the versions from the first build until the Dockerfile changes at or above that step or the image is rebuilt with `--no-cache`. They are apt packages owned by root, so they cannot be upgraded from a session; `az extension update --name azure-devops` is the exception and works without a rebuild.
|
||||
|
||||
## Local customisation
|
||||
|
||||
Compose merges `docker-compose.override.yml` on top of `docker-compose.yaml`. Keep host-specific changes there rather than editing `docker-compose.yaml`, so this repository can be updated without losing them. Both `docker-compose.override.yml` and `docker-compose.override.yaml` are ignored by Git.
|
||||
|
||||
@@ -0,0 +1,255 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# The scripted major-update path for the Docker Compose deployment.
|
||||
#
|
||||
# docker/README.md and docs/docker-self-update.md both point operators here for
|
||||
# anything the in-app updater itself refuses to apply: a changed
|
||||
# `server.Dockerfile`, a changed `docker-compose.yaml`, or a new required
|
||||
# `.env.example` key. None of those can be applied by a container restarting
|
||||
# itself — a restart reuses the existing image and configuration (see "The
|
||||
# environment gate" in docs/docker-self-update.md) — so this script does the
|
||||
# three things an in-place update cannot: force a real image rebuild with no
|
||||
# layer cache, stop the stack, then hand off to Start-Codeman.sh for the same
|
||||
# careful PUID/PGID, override-file and fingerprint handling every other start
|
||||
# goes through.
|
||||
#
|
||||
# ⚠️ Build BEFORE stopping the stack, deliberately, same reasoning as
|
||||
# Start-Codeman.sh's own build-then-down ordering: the build needs nothing
|
||||
# stopped, so a slow --no-cache rebuild costs no downtime, and a build failure
|
||||
# (a bad Dockerfile edit, a network blip pulling a base image) leaves the
|
||||
# ALREADY-RUNNING stack untouched instead of stopped with nothing to bring it
|
||||
# back.
|
||||
#
|
||||
# ⚠️ Clears the codeman-node-modules/codeman-dist named volumes by DEFAULT.
|
||||
# Docker seeds a named volume from the image only while that volume is EMPTY,
|
||||
# so a rebuilt image's fresh node_modules/dist otherwise sit unused behind a
|
||||
# volume's old content and the container comes back up looking unchanged —
|
||||
# exactly wrong for a script whose whole point is "be certain of what ships".
|
||||
# Start-Codeman.sh clears codeman-dist when the checkout's HEAD moved and
|
||||
# codeman-node-modules only when `package-lock.json` changed. A released
|
||||
# server.Dockerfile change arrives through `git pull`, so HEAD moves and dist
|
||||
# is refreshed, but a Dockerfile change that bumps the Node base image leaves
|
||||
# the lockfile untouched while every native module (node-pty is compiled from
|
||||
# source, there is no Linux prebuild) has to be rebuilt against the new Node
|
||||
# ABI. Start-Codeman.sh would keep the old codeman-node-modules volume, and it
|
||||
# never builds with --no-cache. This script clears BOTH volumes, and ONLY
|
||||
# those two (targeted `docker volume rm` by Compose label, never
|
||||
# `down --volumes`, which would also take any volume an override file adds).
|
||||
# Pass --keep-volumes to opt out and reuse whatever is already in them.
|
||||
#
|
||||
# Usage: docker/Update-Codeman.sh [--keep-volumes]
|
||||
# --keep-volumes Do not clear codeman-node-modules/codeman-dist. Safe to
|
||||
# combine with a source change Start-Codeman.sh's own
|
||||
# detection would have cleared anyway; unsafe if the reason
|
||||
# you are here is a change to server.Dockerfile alone.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
script_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)
|
||||
env_file="$script_dir/.env"
|
||||
compose_file="$script_dir/docker-compose.yaml"
|
||||
|
||||
keep_volumes=0
|
||||
for arg in "$@"; do
|
||||
case "$arg" in
|
||||
--keep-volumes)
|
||||
keep_volumes=1
|
||||
;;
|
||||
--help | -h)
|
||||
printf 'Usage: bash %s [--keep-volumes]\n' "$0"
|
||||
exit 0
|
||||
;;
|
||||
*)
|
||||
printf 'Error: unrecognised argument: %s\n' "$arg" >&2
|
||||
printf 'Usage: bash %s [--keep-volumes]\n' "$0" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
if [[ ! -f "$env_file" ]]; then
|
||||
printf 'Error: Docker environment file is missing: %s\n' "$env_file" >&2
|
||||
printf 'Create it from %s/.env.example before running this script.\n' "$script_dir" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Same override-file discovery as Start-Codeman.sh, and deliberately kept in
|
||||
# step with it: a stack built here and started there must resolve to the exact
|
||||
# same Compose files, or this script's build could target a configuration the
|
||||
# handoff's own `up` never actually uses. Compose's own precedence (measured on
|
||||
# v5.5.0 with both present: it uses .yml and ignores .yaml).
|
||||
override_yml="$script_dir/docker-compose.override.yml"
|
||||
override_yaml="$script_dir/docker-compose.override.yaml"
|
||||
if [[ -f "$override_yml" && -f "$override_yaml" ]]; then
|
||||
printf 'Warning: both %s and %s exist; Compose uses .yml and ignores .yaml.\n' \
|
||||
"$override_yml" "$override_yaml" >&2
|
||||
fi
|
||||
compose_files=(-f "$compose_file")
|
||||
for override_file in "$override_yml" "$override_yaml"; do
|
||||
if [[ -f "$override_file" ]]; then
|
||||
compose_files+=(-f "$override_file")
|
||||
printf 'Using Compose override file: %s\n' "$override_file"
|
||||
break
|
||||
fi
|
||||
done
|
||||
compose_command=(docker compose --env-file "$env_file" "${compose_files[@]}")
|
||||
|
||||
# Collision guard. Start-Codeman.sh has no equivalent; this is the only one,
|
||||
# and it has to run before this script's own --no-cache build, `down` and
|
||||
# volume removal below. docker-compose.yaml hard-codes `name: codeman`, so a
|
||||
# second checkout run without COMPOSE_PROJECT_NAME resolves to the SAME Compose
|
||||
# project as any other checkout on the host and would operate on ITS
|
||||
# containers and volumes.
|
||||
#
|
||||
# The project name is read from the resolved config's top-level `name` key
|
||||
# (the first `name` in the output; nested ones come later), the same parse
|
||||
# Start-Codeman.sh uses. `--format json` needs Compose v2.3+. This is the first
|
||||
# `docker` call the script makes, so its failure is reported here rather than
|
||||
# left to `set -e`, which would exit with no output at all.
|
||||
if ! project_config=$("${compose_command[@]}" config --format json); then
|
||||
printf 'Error: `docker compose config --format json` failed (see the message above, if any).\n' >&2
|
||||
printf 'Check that Docker and Compose v2.3+ are installed and on PATH, and that\n' >&2
|
||||
printf '%s and the Compose files in %s are valid.\n' "$env_file" "$script_dir" >&2
|
||||
exit 1
|
||||
fi
|
||||
project_name=$(
|
||||
printf '%s\n' "$project_config" |
|
||||
sed -n 's/^[[:space:]]*"name":[[:space:]]*"\([^"]*\)".*$/\1/p' | head -n1
|
||||
)
|
||||
if [[ -n "$project_name" ]]; then
|
||||
# `|| true` on the pipeline's LAST command: under `set -o pipefail`, `grep -v`
|
||||
# exits 1 when nothing survives the filter — the ordinary, no-collision case,
|
||||
# since `docker ps` finds nothing at all on a first-ever deployment or a
|
||||
# single matching (own) working_dir gets filtered out. Without it, that exit
|
||||
# status propagates through the command substitution and `set -e` aborts the
|
||||
# WHOLE script right here, every time, regardless of whether a collision
|
||||
# actually exists — caught only by actually running this end-to-end (a
|
||||
# static text/regex check on the source cannot see it). The empty-line
|
||||
# filter keeps a container with no working_dir label from winning head -n1
|
||||
# and hiding a real collision behind it.
|
||||
other_working_dir=$(
|
||||
docker ps -a --filter "label=com.docker.compose.project=$project_name" \
|
||||
--format '{{.Label "com.docker.compose.project.working_dir"}}' 2>/dev/null |
|
||||
grep -v -F -x -- "$script_dir" | grep -v '^$' | head -n1 || true
|
||||
)
|
||||
if [[ -n "$other_working_dir" ]]; then
|
||||
printf 'Error: Compose project "%s" is already in use by a DIFFERENT checkout:\n' "$project_name" >&2
|
||||
printf ' %s\n' "$other_working_dir" >&2
|
||||
printf 'This checkout is:\n' >&2
|
||||
printf ' %s\n' "$script_dir" >&2
|
||||
printf '\n' >&2
|
||||
printf 'docker-compose.yaml hard-codes `name: %s`, so two checkouts on the same host\n' "$project_name" >&2
|
||||
printf 'collide unless each one sets a distinct COMPOSE_PROJECT_NAME. Continuing would\n' >&2
|
||||
printf 'rebuild and stop the OTHER checkout'"'"'s running container and, by default,\n' >&2
|
||||
printf 'delete its codeman-node-modules/codeman-dist volumes.\n' >&2
|
||||
printf '\n' >&2
|
||||
printf 'Fix: export COMPOSE_PROJECT_NAME=<something-unique-to-this-checkout> before\n' >&2
|
||||
printf 'running this script, then retry.\n' >&2
|
||||
printf '\n' >&2
|
||||
printf 'If instead THIS checkout was moved or renamed after its container was created,\n' >&2
|
||||
printf 'the path above is its own old location: remove the old container (for example\n' >&2
|
||||
printf '`docker rm -f <container>` for the codeman container) and retry, rather than\n' >&2
|
||||
printf 'setting COMPOSE_PROJECT_NAME, which would start a second project beside it.\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
# Same owner-detection Start-Codeman.sh uses to derive PUID/PGID for its own
|
||||
# build — without it, the --no-cache build below gets Compose's untouched
|
||||
# default of 1000:1000, and on any host whose appdata owner differs (99:100 on
|
||||
# Unraid, per docker/README.md's chown example), Start-Codeman.sh's own
|
||||
# correctly-PUID'd build during the handoff then rebuilds those layers with the
|
||||
# right values anyway — so the "no cache, certain of what ships" image this
|
||||
# script produces is not the one that actually ends up running.
|
||||
#
|
||||
# Deliberately NOT the same as Start-Codeman.sh's own handling of a MISSING
|
||||
# appdata directory (which creates it): this script updates an EXISTING
|
||||
# deployment, so a missing appdata path means there is nothing here yet to
|
||||
# update, and creating one would just be this script quietly doing
|
||||
# Start-Codeman.sh's first-run job worse.
|
||||
appdata_path=$(
|
||||
"${compose_command[@]}" config --environment |
|
||||
awk -F= '$1 == "CODEMAN_APPDATA_PATH" { sub(/^[^=]*=/, ""); print; exit }'
|
||||
)
|
||||
if [[ -z "$appdata_path" || ! -d "$appdata_path" ]]; then
|
||||
printf 'Error: CODEMAN_APPDATA_PATH is not set or does not exist: %s\n' "${appdata_path:-<unset>}" >&2
|
||||
printf 'Run docker/Start-Codeman.sh first to set up a new deployment.\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# `stat -c` is GNU, `stat -f` is BSD/macOS; the bind source lives on the Docker
|
||||
# host, so both need to work. Identical to Start-Codeman.sh's own helper.
|
||||
owner_of() {
|
||||
stat -c '%u:%g' -- "$1" 2>/dev/null || stat -f '%u:%g' "$1" 2>/dev/null
|
||||
}
|
||||
|
||||
if ! owner_ids=$(owner_of "$appdata_path"); then
|
||||
printf 'Error: Cannot determine the owner of CODEMAN_APPDATA_PATH: %s\n' "$appdata_path" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
export PUID=${owner_ids%%:*}
|
||||
export PGID=${owner_ids##*:}
|
||||
|
||||
if [[ "$PUID" == '0' ]]; then
|
||||
printf 'Error: CODEMAN_APPDATA_PATH is owned by root: %s\n' "$appdata_path" >&2
|
||||
printf 'Change the directory ownership to the unprivileged account that should run Codeman.\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# --no-cache, always: a plain `build` reuses cached layers (npm install, apt
|
||||
# packages, the CLI installs baked into the image) and can silently keep them
|
||||
# frozen at whatever they were the day the cache was populated — exactly wrong
|
||||
# for a major update, whose whole point is being certain of what actually
|
||||
# ships. `scripts/build-agent-image.mjs` makes the same call for the same
|
||||
# reason (see its entry in CLAUDE.md's Additional Commands table). Runs BEFORE
|
||||
# the stack is stopped — see the header comment for why.
|
||||
printf 'Building a fresh image (--no-cache)...\n'
|
||||
"${compose_command[@]}" build --no-cache
|
||||
|
||||
printf 'Stopping the stack...\n'
|
||||
if [[ "$keep_volumes" == '1' || -n "$project_name" ]]; then
|
||||
"${compose_command[@]}" down
|
||||
else
|
||||
# No resolvable project name means the label filter below could match
|
||||
# nothing, so fall back to Compose's own removal, and say what it really does.
|
||||
printf 'Warning: could not resolve the Compose project name; clearing EVERY named volume\n' >&2
|
||||
printf 'in this Compose project (override file included) with `down --volumes` instead.\n' >&2
|
||||
"${compose_command[@]}" down --volumes
|
||||
fi
|
||||
|
||||
# Targeted removal of exactly the two build-artefact volumes, scoped by label to
|
||||
# THIS project (the volume key alone is shared by any other stack declaring the
|
||||
# same key). Same lookup as Start-Codeman.sh's refresh. A failure is reported,
|
||||
# not fatal: the stack is already down, and the handoff below is what brings
|
||||
# it back up.
|
||||
if [[ "$keep_volumes" != '1' && -n "$project_name" ]]; then
|
||||
printf 'Clearing the codeman-node-modules/codeman-dist volumes (pass --keep-volumes to skip).\n'
|
||||
for key in codeman-node-modules codeman-dist; do
|
||||
volume_name=$(
|
||||
docker volume ls -q \
|
||||
--filter "label=com.docker.compose.volume=$key" \
|
||||
--filter "label=com.docker.compose.project=$project_name" |
|
||||
head -n1
|
||||
) || volume_name=''
|
||||
if [[ -n "$volume_name" ]] && ! docker volume rm -- "$volume_name"; then
|
||||
printf 'Warning: could not remove volume %s; the container may keep serving the\n' "$volume_name" >&2
|
||||
printf 'previous build from it. Remove it by hand and rerun this script.\n' >&2
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
# Start-Codeman.sh does everything a plain `up -d` does not: re-derives
|
||||
# PUID/PGID, pre-creates CODEMAN_CASES_PATH with the right ownership, resolves
|
||||
# DOCKER_SOCKET_GID, records the server.Dockerfile/docker-compose.yaml
|
||||
# fingerprint the in-app updater's gate reads on every future update, and
|
||||
# starts the (already freshly built) image. Reimplementing any of that here
|
||||
# would only risk drifting out of step with it — hand off instead, exactly as
|
||||
# docs/docker-self-update.md's own reset procedure does.
|
||||
#
|
||||
# ⚠️ `bash`, not a bare exec of the path: Start-Codeman.sh is committed
|
||||
# non-executable (100644), the same as this script, and is documented
|
||||
# everywhere as `bash docker/Start-Codeman.sh` rather than
|
||||
# `./docker/Start-Codeman.sh` — execing the bare path fails with EACCES.
|
||||
printf 'Handing off to Start-Codeman.sh...\n'
|
||||
exec bash "$script_dir/Start-Codeman.sh"
|
||||
@@ -17,6 +17,7 @@ FROM node:22-bookworm-slim
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends \
|
||||
git \
|
||||
libsecret-1-0 \
|
||||
tmux \
|
||||
ripgrep \
|
||||
curl \
|
||||
@@ -26,6 +27,88 @@ RUN apt-get update \
|
||||
openssh-client \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# GitHub CLI and Azure CLI (+ the azure-devops extension) with the same system
|
||||
# git credential helpers as docker/server.Dockerfile, so an agent in a Docker
|
||||
# case can clone and push to private GitHub / Azure DevOps repositories. The
|
||||
# sign-ins themselves are NOT baked in: `~/.config/gh` and `~/.azure` are seeded
|
||||
# per container at launch like every other CLI's credentials (CRED_STORES in
|
||||
# src/docker-hosts.ts), and a helper whose CLI is not signed in prints nothing,
|
||||
# so git fails fast instead of prompting. See server.Dockerfile for why the
|
||||
# vendor apt repositories are configured here rather than via deb_install.sh.
|
||||
#
|
||||
# Each is OPT-IN and OFF by default, like the server image: CODEMAN_INSTALL_GH=1
|
||||
# / CODEMAN_INSTALL_AZ=1 turn one on; off leaves no repository, package,
|
||||
# extension or helper entry. scripts/build-agent-image.mjs and the in-app
|
||||
# auto-build pass them from CODEMAN_AGENT_IMAGE_INSTALL_GH / _AZ in their own
|
||||
# environment (for the Compose deployment: `environment:` in
|
||||
# docker-compose.override.yml), and pass nothing when those are unset, so
|
||||
# these defaults (off) apply.
|
||||
ARG CODEMAN_INSTALL_GH=0
|
||||
ARG CODEMAN_INSTALL_AZ=0
|
||||
RUN set -eux; \
|
||||
for flag in "CODEMAN_INSTALL_GH=${CODEMAN_INSTALL_GH}" "CODEMAN_INSTALL_AZ=${CODEMAN_INSTALL_AZ}"; do \
|
||||
case "${flag#*=}" in 0|1) ;; *) echo "${flag%%=*} must be 0 or 1, got '${flag#*=}'" >&2; exit 1;; esac; \
|
||||
done; \
|
||||
codename="$(. /etc/os-release && echo "${VERSION_CODENAME}")"; \
|
||||
arch="$(dpkg --print-architecture)"; \
|
||||
pkgs=""; \
|
||||
install -d -m 0755 /etc/apt/keyrings; \
|
||||
if [ "${CODEMAN_INSTALL_GH}" = 1 ]; then \
|
||||
curl -fsSL -o /etc/apt/keyrings/githubcli-archive-keyring.gpg \
|
||||
https://cli.github.com/packages/githubcli-archive-keyring.gpg; \
|
||||
chmod go+r /etc/apt/keyrings/githubcli-archive-keyring.gpg; \
|
||||
echo "deb [arch=${arch} signed-by=/etc/apt/keyrings/githubcli-archive-keyring.gpg] https://cli.github.com/packages stable main" \
|
||||
> /etc/apt/sources.list.d/github-cli.list; \
|
||||
pkgs="${pkgs} gh"; \
|
||||
fi; \
|
||||
if [ "${CODEMAN_INSTALL_AZ}" = 1 ]; then \
|
||||
curl -fsSL -o /etc/apt/keyrings/microsoft.asc \
|
||||
https://packages.microsoft.com/keys/microsoft.asc; \
|
||||
chmod go+r /etc/apt/keyrings/microsoft.asc; \
|
||||
echo "deb [arch=${arch} signed-by=/etc/apt/keyrings/microsoft.asc] https://packages.microsoft.com/repos/azure-cli/ ${codename} main" \
|
||||
> /etc/apt/sources.list.d/azure-cli.list; \
|
||||
pkgs="${pkgs} azure-cli"; \
|
||||
fi; \
|
||||
if [ -n "${pkgs}" ]; then \
|
||||
apt-get update; \
|
||||
apt-get install -y --no-install-recommends ${pkgs}; \
|
||||
rm -rf /var/lib/apt/lists/*; \
|
||||
fi; \
|
||||
if [ "${CODEMAN_INSTALL_GH}" = 1 ]; then gh --version; fi; \
|
||||
if [ "${CODEMAN_INSTALL_AZ}" = 1 ]; then az version --output none; fi
|
||||
|
||||
# Outside HOME so the seeded `~/.azure` (auth files only) never has to carry
|
||||
# extensions. gid 0 + group-writable, the same arbitrary-uid convention as HOME
|
||||
# below, so `az extension update` works as whatever uid the container runs as.
|
||||
# Created even without az; an empty directory costs nothing.
|
||||
ENV AZURE_EXTENSION_DIR=/opt/az-extensions
|
||||
RUN set -eux; \
|
||||
install -d -m 0755 "${AZURE_EXTENSION_DIR}"; \
|
||||
if [ "${CODEMAN_INSTALL_AZ}" = 1 ]; then \
|
||||
az extension add --name azure-devops --only-show-errors; \
|
||||
rm -rf /root/.azure; \
|
||||
fi; \
|
||||
chgrp -R 0 "${AZURE_EXTENSION_DIR}"; \
|
||||
chmod -R g=u "${AZURE_EXTENSION_DIR}"
|
||||
|
||||
# Only an installed CLI gets a helper entry (see server.Dockerfile).
|
||||
COPY docker/git-credential-azure-cli /usr/local/bin/git-credential-azure-cli
|
||||
RUN set -eux; \
|
||||
if [ "${CODEMAN_INSTALL_GH}" = 1 ]; then \
|
||||
for host in https://github.com https://gist.github.com; do \
|
||||
git config --system "credential.${host}.helper" '!/usr/bin/gh auth git-credential'; \
|
||||
done; \
|
||||
fi; \
|
||||
if [ "${CODEMAN_INSTALL_AZ}" = 1 ]; then \
|
||||
chmod 0755 /usr/local/bin/git-credential-azure-cli; \
|
||||
for host in https://dev.azure.com 'https://*.visualstudio.com'; do \
|
||||
git config --system "credential.${host}.helper" /usr/local/bin/git-credential-azure-cli; \
|
||||
git config --system "credential.${host}.useHttpPath" true; \
|
||||
done; \
|
||||
else \
|
||||
rm -f /usr/local/bin/git-credential-azure-cli; \
|
||||
fi
|
||||
|
||||
# The npm-published agent CLIs, supplied by scripts/build-agent-image.mjs from
|
||||
# config/clis.stock.json so a new stock CLI needs no edit here. The default is
|
||||
# today's literal list, so a bare `docker build` still produces the same image.
|
||||
@@ -44,6 +127,10 @@ RUN apt-get update \
|
||||
# A different order is a different RUN string, which is a different layer hash and
|
||||
# so a needless cache miss between a bare `docker build` and a scripted one.
|
||||
ARG CLI_NPM_PACKAGES="@anthropic-ai/claude-code opencode-ai @openai/codex @google/gemini-cli"
|
||||
# uv/uvx: MCP servers are commonly launched with `uvx <package>` (e.g. the Nginx
|
||||
# Proxy Manager MCP), and Codex failed to enable them with "uvx not found". Copied
|
||||
# from the pinned upstream image into root-owned /usr/local/bin, never pip-installed.
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.9 /uv /uvx /usr/local/bin/
|
||||
RUN npm install -g ${CLI_NPM_PACKAGES} \
|
||||
&& npm cache clean --force
|
||||
|
||||
@@ -167,6 +254,21 @@ RUN useradd -g 0 -m -d /home/agent -s /bin/bash agent \
|
||||
&& chgrp -R 0 /home/agent \
|
||||
&& chmod -R g=u /home/agent
|
||||
|
||||
# Docker cases have a fresh, container-owned home directory. Declare the
|
||||
# optional identity here so changing it invalidates only this final layer, then
|
||||
# configure Git's system defaults. A user-level config still takes precedence.
|
||||
ARG GIT_USER_EMAIL=
|
||||
ARG GIT_USER_NAME=
|
||||
RUN set -eux; \
|
||||
if [ -n "${GIT_USER_NAME}" ] || [ -n "${GIT_USER_EMAIL}" ]; then \
|
||||
if [ -z "${GIT_USER_NAME}" ] || [ -z "${GIT_USER_EMAIL}" ]; then \
|
||||
echo 'Git user name and email must both be set when configuring Git identity' >&2; \
|
||||
exit 1; \
|
||||
fi; \
|
||||
git config --system user.name "${GIT_USER_NAME}"; \
|
||||
git config --system user.email "${GIT_USER_EMAIL}"; \
|
||||
fi
|
||||
|
||||
USER agent
|
||||
WORKDIR /home/agent
|
||||
|
||||
|
||||
@@ -7,6 +7,8 @@ services:
|
||||
dockerfile: docker/server.Dockerfile
|
||||
args:
|
||||
CODEMAN_RUNTIME_USER: ${CODEMAN_RUNTIME_USER}
|
||||
GIT_USER_EMAIL: ${GIT_USER_EMAIL:-}
|
||||
GIT_USER_NAME: ${GIT_USER_NAME:-}
|
||||
PGID: ${PGID:-1000}
|
||||
PUID: ${PUID:-1000}
|
||||
image: ${CODEMAN_IMAGE}
|
||||
@@ -32,6 +34,10 @@ services:
|
||||
CODEMAN_DOCKER_HOST_HOME: ${CODEMAN_APPDATA_PATH}
|
||||
CODEMAN_DOCKER_DISABLE_SWAP_LIMIT: ${CODEMAN_DOCKER_DISABLE_SWAP_LIMIT}
|
||||
CODEMAN_CASES_PATH: ${CODEMAN_CASES_PATH}
|
||||
# Passed through only so Codeman can use the same identity when it builds
|
||||
# the Docker-case agent image.
|
||||
CODEMAN_AGENT_IMAGE_GIT_USER_EMAIL: ${GIT_USER_EMAIL:-}
|
||||
CODEMAN_AGENT_IMAGE_GIT_USER_NAME: ${GIT_USER_NAME:-}
|
||||
# Extra Host-header allowlist entries for a reverse-proxied deployment
|
||||
# (docker/README.md, "Reverse-proxy host allowlist"). Optional, so it
|
||||
# defaults to empty rather than requiring a line in every .env.
|
||||
|
||||
Executable
+34
@@ -0,0 +1,34 @@
|
||||
#!/bin/sh
|
||||
# Git credential helper for Azure DevOps, backed by the signed-in Azure CLI.
|
||||
#
|
||||
# Configured in the image's system gitconfig for https://dev.azure.com and
|
||||
# https://*.visualstudio.com (see server.Dockerfile). On `get` it answers with
|
||||
# an Entra ID access token for the Azure DevOps resource as the password, the
|
||||
# same token type Git Credential Manager uses for Azure Repos. It never prompts:
|
||||
# when `az` is not signed in it prints nothing, so git fails fast with its own
|
||||
# authentication error instead of hanging a request that has no terminal.
|
||||
#
|
||||
# AZURE_DEVOPS_EXT_PAT, the azure-devops extension's own PAT variable, is used
|
||||
# instead when it is set, for accounts that authenticate with a PAT.
|
||||
|
||||
# `store` and `erase` are no-ops: the token belongs to az, which refreshes it.
|
||||
[ "$1" = "get" ] || exit 0
|
||||
|
||||
# Drain the request git writes on stdin; the host scoping is in gitconfig.
|
||||
cat >/dev/null
|
||||
|
||||
if [ -n "${AZURE_DEVOPS_EXT_PAT:-}" ]; then
|
||||
printf 'username=pat\npassword=%s\n' "$AZURE_DEVOPS_EXT_PAT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
command -v az >/dev/null 2>&1 || exit 0
|
||||
|
||||
# 499b84ac-1321-427f-aa17-267ca6975798 is the fixed application ID of Azure
|
||||
# DevOps: https://learn.microsoft.com/azure/devops/integrate/get-started/authentication/service-principal-managed-identity
|
||||
token="$(az account get-access-token \
|
||||
--resource 499b84ac-1321-427f-aa17-267ca6975798 \
|
||||
--query accessToken --output tsv 2>/dev/null)" || exit 0
|
||||
[ -n "$token" ] || exit 0
|
||||
|
||||
printf 'username=azure-cli\npassword=%s\n' "$token"
|
||||
+137
-1
@@ -39,6 +39,7 @@ RUN apt-get update \
|
||||
curl \
|
||||
g++ \
|
||||
git \
|
||||
libsecret-1-0 \
|
||||
make \
|
||||
openssh-client \
|
||||
procps \
|
||||
@@ -68,6 +69,111 @@ COPY --from=docker:29-cli \
|
||||
/usr/local/libexec/docker/cli-plugins/docker-buildx \
|
||||
/usr/local/libexec/docker/cli-plugins/docker-buildx
|
||||
|
||||
# GitHub CLI and Azure CLI (with the azure-devops extension), so a user can sign
|
||||
# this container in to GitHub and Azure DevOps from a Codeman shell session and
|
||||
# then clone PRIVATE repositories, both from that session and through Add Case
|
||||
# -> Clone Repo. Codeman still collects no Git credentials itself: the clone
|
||||
# path (src/git-clone.ts) only inherits HOME and git's config, so whatever the
|
||||
# user signs in to here is what authenticates, and nothing when they have not
|
||||
# (the clone then fails fast with AUTH_REQUIRED, exactly as before).
|
||||
#
|
||||
# Each is OPT-IN and OFF by default: the image is functionally unchanged
|
||||
# unless the build gets CODEMAN_INSTALL_GH=1 and/or CODEMAN_INSTALL_AZ=1, which
|
||||
# a deployment sets under `build: args:` in docker-compose.override.yml
|
||||
# (docker/README.md, "Private repositories"). Off installs no apt repository,
|
||||
# package, extension or credential-helper entry; all that remains is the
|
||||
# AZURE_EXTENSION_DIR variable, its empty directory and one layer that copies
|
||||
# and then removes the helper script. The Azure CLI is the heavy one (~600 MB,
|
||||
# mostly its bundled Python). The base docker-compose.yaml
|
||||
# and .env deliberately do not carry them: turning a CLI on is a per-host
|
||||
# choice, which is what the override file is for, and a new .env.example key
|
||||
# would make the self-updater refuse existing installs until their .env gained
|
||||
# it (docs/docker-self-update.md).
|
||||
#
|
||||
# Both come from their vendors' own apt repositories, the same ones the
|
||||
# documented one-liners configure (https://github.com/cli/cli/blob/trunk/docs/install_linux.md
|
||||
# and https://learn.microsoft.com/cli/azure/install-azure-cli-linux?pivots=apt).
|
||||
# Microsoft's `deb_install.sh` is deliberately not piped into the build: it does
|
||||
# exactly this plus a `gnupg` install, and a remote script run at build time is
|
||||
# the one step a reviewer cannot read in this file. apt reads an ASCII-armoured
|
||||
# `.asc` key directly, which is what keeps `gnupg` out of the image.
|
||||
#
|
||||
# Not pinned, unlike the agent CLIs below: nothing in Codeman depends on a
|
||||
# particular gh or az behaviour, so the pinning argument there does not apply.
|
||||
# The layer cache still keeps whatever version the first build fetched until a
|
||||
# --no-cache rebuild.
|
||||
ARG CODEMAN_INSTALL_GH=0
|
||||
ARG CODEMAN_INSTALL_AZ=0
|
||||
RUN set -eux; \
|
||||
for flag in "CODEMAN_INSTALL_GH=${CODEMAN_INSTALL_GH}" "CODEMAN_INSTALL_AZ=${CODEMAN_INSTALL_AZ}"; do \
|
||||
case "${flag#*=}" in 0|1) ;; *) echo "${flag%%=*} must be 0 or 1, got '${flag#*=}'" >&2; exit 1;; esac; \
|
||||
done; \
|
||||
codename="$(. /etc/os-release && echo "${VERSION_CODENAME}")"; \
|
||||
arch="$(dpkg --print-architecture)"; \
|
||||
pkgs=""; \
|
||||
install -d -m 0755 /etc/apt/keyrings; \
|
||||
if [ "${CODEMAN_INSTALL_GH}" = 1 ]; then \
|
||||
curl -fsSL -o /etc/apt/keyrings/githubcli-archive-keyring.gpg \
|
||||
https://cli.github.com/packages/githubcli-archive-keyring.gpg; \
|
||||
chmod go+r /etc/apt/keyrings/githubcli-archive-keyring.gpg; \
|
||||
echo "deb [arch=${arch} signed-by=/etc/apt/keyrings/githubcli-archive-keyring.gpg] https://cli.github.com/packages stable main" \
|
||||
> /etc/apt/sources.list.d/github-cli.list; \
|
||||
pkgs="${pkgs} gh"; \
|
||||
fi; \
|
||||
if [ "${CODEMAN_INSTALL_AZ}" = 1 ]; then \
|
||||
curl -fsSL -o /etc/apt/keyrings/microsoft.asc \
|
||||
https://packages.microsoft.com/keys/microsoft.asc; \
|
||||
chmod go+r /etc/apt/keyrings/microsoft.asc; \
|
||||
echo "deb [arch=${arch} signed-by=/etc/apt/keyrings/microsoft.asc] https://packages.microsoft.com/repos/azure-cli/ ${codename} main" \
|
||||
> /etc/apt/sources.list.d/azure-cli.list; \
|
||||
pkgs="${pkgs} azure-cli"; \
|
||||
fi; \
|
||||
if [ -n "${pkgs}" ]; then \
|
||||
apt-get update; \
|
||||
apt-get install -y --no-install-recommends ${pkgs}; \
|
||||
rm -rf /var/lib/apt/lists/*; \
|
||||
fi
|
||||
|
||||
# The azure-devops extension goes into a SYSTEM directory rather than the
|
||||
# default ~/.azure/cliextensions: HOME is the application-data bind mount, which
|
||||
# hides anything installed there at build time. The directory is handed to the
|
||||
# runtime account below (next to /opt/codeman-cli) so `az extension update`
|
||||
# works from a session. Nothing that runs as root executes from it. It is
|
||||
# created even without az, so the chown below does not have to know.
|
||||
ENV AZURE_EXTENSION_DIR=/opt/codeman-az-extensions
|
||||
RUN set -eux; \
|
||||
install -d -m 0755 "${AZURE_EXTENSION_DIR}"; \
|
||||
if [ "${CODEMAN_INSTALL_AZ}" = 1 ]; then \
|
||||
az extension add --name azure-devops --only-show-errors; \
|
||||
rm -rf /root/.azure; \
|
||||
fi
|
||||
|
||||
# Git credential helpers, in the SYSTEM gitconfig so they apply to every
|
||||
# account and survive a fresh application-data directory. Each one answers only
|
||||
# for its own host and prints nothing when its CLI is not signed in, so git
|
||||
# falls through to its normal non-interactive failure. Only an installed CLI
|
||||
# gets an entry: a helper naming a missing binary would print an error on every
|
||||
# clone from that host.
|
||||
# github.com `gh auth git-credential`, what `gh auth setup-git` configures.
|
||||
# Azure DevOps an Entra ID token from `az login` (git-credential-azure-cli),
|
||||
# for both dev.azure.com and the legacy *.visualstudio.com hosts.
|
||||
COPY docker/git-credential-azure-cli /usr/local/bin/git-credential-azure-cli
|
||||
RUN set -eux; \
|
||||
if [ "${CODEMAN_INSTALL_GH}" = 1 ]; then \
|
||||
for host in https://github.com https://gist.github.com; do \
|
||||
git config --system "credential.${host}.helper" '!/usr/bin/gh auth git-credential'; \
|
||||
done; \
|
||||
fi; \
|
||||
if [ "${CODEMAN_INSTALL_AZ}" = 1 ]; then \
|
||||
chmod 0755 /usr/local/bin/git-credential-azure-cli; \
|
||||
for host in https://dev.azure.com 'https://*.visualstudio.com'; do \
|
||||
git config --system "credential.${host}.helper" /usr/local/bin/git-credential-azure-cli; \
|
||||
git config --system "credential.${host}.useHttpPath" true; \
|
||||
done; \
|
||||
else \
|
||||
rm -f /usr/local/bin/git-credential-azure-cli; \
|
||||
fi
|
||||
|
||||
# Keep credentials out of the image. Users authenticate these CLIs at runtime
|
||||
# through Codeman sessions, and the configured host bind mount retains state.
|
||||
#
|
||||
@@ -107,13 +213,28 @@ COPY --from=docker:29-cli \
|
||||
# minimal image of this exact shape). The four CLIs live only in this prefix,
|
||||
# so they still resolve; entrypoint.sh additionally pins its own PATH to the
|
||||
# system directories for the root part of the start.
|
||||
# uv/uvx: MCP servers are commonly launched with `uvx <package>` (e.g. the Nginx
|
||||
# Proxy Manager MCP), and Codex failed to enable them with "uvx not found". Copied
|
||||
# from the pinned upstream image into root-owned /usr/local/bin, never pip-installed.
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.9 /uv /uvx /usr/local/bin/
|
||||
ENV NPM_CONFIG_PREFIX=/opt/codeman-cli
|
||||
ENV PATH=$PATH:/opt/codeman-cli/bin
|
||||
# CLIs installed at runtime (Settings -> CLIs, npm redirected to ~/.local by installEnv()) live on the
|
||||
# persistent home mount, so they survive a container recreate. Appended for the same reason as above.
|
||||
ENV PATH=$PATH:/home/${CODEMAN_RUNTIME_USER}/.local/bin
|
||||
# pnpm is not an agent CLI: it is here because `dsh plugin` (DeepSeek Harness, which
|
||||
# this image leaves to be installed at runtime, see SERVER_INTENTIONAL_OMISSIONS in
|
||||
# test/docker-agent-image-coverage.test.ts) spawns a literal `pnpm` with no npm
|
||||
# fallback, so the Run menu's "DeepSeek - add a terminal profile" button failed
|
||||
# with `dsh: pnpm not found on PATH` (exit 127) on this image. The agent image
|
||||
# already carries it for the same reason (#352). It lives in the same
|
||||
# runtime-writable prefix as the CLIs, so a session can update it in place.
|
||||
RUN npm install --global \
|
||||
@anthropic-ai/claude-code@2.1.258 \
|
||||
@google/gemini-cli@0.58.0 \
|
||||
@openai/codex@0.152.1 \
|
||||
opencode-ai@1.18.26 \
|
||||
pnpm@12.6.0 \
|
||||
&& npm cache clean --force
|
||||
|
||||
# Keep the web server and every local Codeman session unprivileged. PUID and
|
||||
@@ -153,7 +274,7 @@ RUN set -eux; \
|
||||
--shell /bin/bash \
|
||||
"${CODEMAN_RUNTIME_USER}"; \
|
||||
fi; \
|
||||
chown -R "${PUID}:${PGID}" /opt/codeman-cli
|
||||
chown -R "${PUID}:${PGID}" /opt/codeman-cli /opt/codeman-az-extensions
|
||||
|
||||
WORKDIR /opt/codeman
|
||||
|
||||
@@ -181,6 +302,21 @@ EXPOSE 3000
|
||||
COPY docker/entrypoint.sh /usr/local/bin/entrypoint.sh
|
||||
RUN chmod 0755 /usr/local/bin/entrypoint.sh
|
||||
|
||||
# Declare the optional identity immediately before configuring it so a change
|
||||
# invalidates only this final layer. This is declarative setup: a persisted
|
||||
# ~/.gitconfig in CODEMAN_APPDATA_PATH still overrides the system-level values.
|
||||
ARG GIT_USER_EMAIL=
|
||||
ARG GIT_USER_NAME=
|
||||
RUN set -eux; \
|
||||
if [ -n "${GIT_USER_NAME}" ] || [ -n "${GIT_USER_EMAIL}" ]; then \
|
||||
if [ -z "${GIT_USER_NAME}" ] || [ -z "${GIT_USER_EMAIL}" ]; then \
|
||||
echo 'Git user name and email must both be set when configuring Git identity' >&2; \
|
||||
exit 1; \
|
||||
fi; \
|
||||
git config --system user.name "${GIT_USER_NAME}"; \
|
||||
git config --system user.email "${GIT_USER_EMAIL}"; \
|
||||
fi
|
||||
|
||||
ENTRYPOINT ["/usr/local/bin/entrypoint.sh"]
|
||||
|
||||
CMD ["node", "dist/index.js", "web"]
|
||||
|
||||
@@ -757,3 +757,13 @@ works, and its replies arrive tagged `from-name="w9-msgtest"` (a derived-name
|
||||
worker's replies carry no `from-name`). A quick-start without `sessionName` has an
|
||||
empty Codeman name, so the peer name stays derived: agents should name their
|
||||
workers. Tests: `test/name-flag-injection.test.ts`.
|
||||
|
||||
Later narrowing: `--name` is not only the peer name but also the `/resume` picker
|
||||
entry and the terminal title, and a pinned title stops Claude generating its own, so
|
||||
pinning the `w1-myapp` placeholder listed every conversation of a case under the same
|
||||
name in `/resume`. Only a manual name is pinned now (`Session.cliPinnedName`,
|
||||
`nameSource === 'manual'`, carried to the builders as `cliName`); placeholder and auto
|
||||
names leave Claude to title the conversation. A rename in Codeman appends a
|
||||
`custom-title` row to the conversation's transcript (`claude-session-title.ts`), the
|
||||
row `/rename` writes. Tests: `test/claude-resume-title.test.ts`,
|
||||
`test/routes/session-name-routes.test.ts`.
|
||||
|
||||
@@ -311,6 +311,15 @@ worker's prompt but never submitted, and the wait then runs its full timeout on
|
||||
turn that never started. Verified live; this is the most common silent failure on
|
||||
this endpoint.
|
||||
|
||||
A **plain prompt** (printable text followed by exactly one `\r`, nothing else) is
|
||||
delivered through tmux even without `useMux`: the text is typed, Enter is pressed as
|
||||
a separate key, and the server re-presses Enter while the prompt is still visibly
|
||||
sitting on the composer. Written straight into the pane in one piece, a prompt of
|
||||
about a hundred characters or more is taken as a paste by Claude Code, its `\r`
|
||||
becomes a newline, and the prompt stays unsent (measured on 2.1.283). Any other
|
||||
input (escape sequences, a bracketed-paste frame, a line feed, a bare `\r`) keeps
|
||||
the raw write, and an explicit `"useMux": false` forces it.
|
||||
|
||||
```bash
|
||||
curl -s -X POST "$API/api/v1/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
@@ -324,6 +333,30 @@ from the session's current state rather than requiring a new transition: the
|
||||
original turn may be long over. It comes back as
|
||||
`"delivered": false, "duplicate": true`.
|
||||
|
||||
**Wake-on-LAN hosts** (`docs/remote-sessions.md` §Wake-on-LAN): when the session's
|
||||
remote host has a wake target and is asleep, the non-wait form answers `200` with
|
||||
`{"buffered": true}` — the bytes are held and flushed after the host is back — or
|
||||
`{"buffered": true, "dropped": true}` for a chunk over the 4 KB wake buffer, which
|
||||
is gone (never delivered as a fragment). Both fields are additive to the historical
|
||||
bare `{}`. With `wait`, the route blocks on the wake instead and answers
|
||||
`422 OPERATION_FAILED` ("did not come back after a wake-on-LAN request — nothing was
|
||||
sent") when the host never returns, rather than writing into the stalled pane and
|
||||
reporting `delivered:true` plus a timeout.
|
||||
|
||||
Two endpoints back that flow directly, both scoped to one session's remote host and
|
||||
both refusing a session that is not remote (`400 INVALID_INPUT`):
|
||||
|
||||
| Method | Path | Purpose |
|
||||
| --- | --- | --- |
|
||||
| `GET` | `/api/sessions/:id/reachability` | Whether the session's remote host answers SSH right now, plus whether a wake target is configured. Read-only: it never wakes. `{"reachable": true\|false\|null, "wakeConfigured": "mac"\|"command"\|"none"}`, where `null` means the answer is unknown (a proxied host, where a TCP probe proves nothing). |
|
||||
| `POST` | `/api/sessions/:id/wake` | Wake the host and wait for it to accept SSH again, bounded by the request budget. `422 OPERATION_FAILED` when it does not come back; `400 INVALID_INPUT` with "No wake-on-LAN target configured for this host" when nothing is set. |
|
||||
|
||||
⚠️ Waking is deliberately reachable only from an explicit user action (this route, a
|
||||
session create/attach, or typing into a sleeping session). No watcher, dropped-session
|
||||
handler or boot-recovery path may wake a host, or a suspended machine would be woken
|
||||
again seconds after every suspend; `test/remote-wake.test.ts` pins that as an import
|
||||
fence around `src/remote-wake.ts`.
|
||||
|
||||
### Response
|
||||
|
||||
All three nest the wait result under `data.wait`, so one client helper works against
|
||||
@@ -558,6 +591,128 @@ All four enforce session ownership in multi-user mode; a foreign session id
|
||||
answers `404 NOT_FOUND` (no existence leak), and profiles of two owners of the
|
||||
same directory are distinct by construction.
|
||||
|
||||
## Custom Model Endpoints
|
||||
|
||||
Points a session's harness at a user-configured OpenAI-compatible endpoint —
|
||||
local (llama.cpp, vLLM, DGX Spark) or cloud (Azure AI Foundry, OpenRouter) —
|
||||
instead of its native cloud backend, gated by the opt-in
|
||||
`customModelEndpointsEnabled` setting (default OFF). Endpoints are
|
||||
machine-level infra, like remote/docker hosts: writes are admin-only in
|
||||
multi-user mode. Design: [`custom-model-endpoints-plan.md`](custom-model-endpoints-plan.md);
|
||||
user guide: [`custom-model-endpoints.md`](custom-model-endpoints.md).
|
||||
|
||||
- `GET /api/v1/model-endpoints` -> `CustomModelHost[]`, an unwrapped bare
|
||||
array like every other list route (still riding the standard `{success,
|
||||
data}` envelope on the wire — unwrap it the same way). Answers `[]` for a
|
||||
non-admin in multi-user mode. `apiKey` is never returned; `apiKeySet:
|
||||
boolean` reports whether one is stored, so a client can render "unchanged
|
||||
if left blank" without ever holding the real value.
|
||||
- `POST /api/v1/model-endpoints` with `{ id, label, baseUrl, apiKey?,
|
||||
authStyle?, defaultModelId? }` creates one. `id` must match
|
||||
`^[a-zA-Z0-9_-]+$`; `authStyle` is `bearer` (default) or `api-key`, never
|
||||
both (a real server hung indefinitely when sent both headers on one
|
||||
request); `baseUrl` must be `http(s)`, carry no embedded credentials, and
|
||||
is refused if it points at (or resolves to) a link-local or
|
||||
cloud-metadata address. `409 ALREADY_EXISTS` on a duplicate id.
|
||||
- `PUT /api/v1/model-endpoints/:id` updates one. An **absent** `apiKey`
|
||||
keeps the stored one rather than clearing it — the client never receives
|
||||
the real value to resend deliberately unchanged, so omission is the only
|
||||
way to say "leave it alone"; there is no way to clear a key back to unset
|
||||
this way. `defaultModelId`, when set, must be one of that endpoint's own
|
||||
`models` (`400 INVALID_INPUT` otherwise).
|
||||
- `DELETE /api/v1/model-endpoints/:id` removes one.
|
||||
- `POST /api/v1/model-endpoints/:id/discover-models` fetches the endpoint's
|
||||
own `GET /v1/models` and stores the result as `models`, updating
|
||||
`lastDiscoveredAt`, plus (best-effort, only for a model llama-swap's own
|
||||
response already reports loaded) `modelContextLengths` and `modelSizesGB`.
|
||||
A `defaultModelId` that no longer appears in the fresh list is dropped
|
||||
rather than carried forward invalid. Failures answer `422 OPERATION_FAILED`
|
||||
with the underlying connection error, or a named egress refusal if the
|
||||
resolved address turned out to be blocked. The same refresh also runs
|
||||
automatically for every saved endpoint every 5 minutes in the background
|
||||
(`refreshAllCustomModelHosts()`, `custom-model-routes.ts`, started from
|
||||
`server.ts`), so there is no route for triggering "refresh all" — one
|
||||
endpoint being unreachable on a cycle never blocks the others.
|
||||
- `GET /api/v1/model-endpoints/:id/running-status` -> `{ isLlamaSwap,
|
||||
running: [{model, state}], logLine? }`, read-only, no admin gate
|
||||
(any session owner who could already point a session at this endpoint can
|
||||
equally ask what it currently has loaded). `isLlamaSwap` is
|
||||
feature-detected via the endpoint's own `GET /running` — a plain
|
||||
llama.cpp/OpenAI-compatible server has none and always answers `false`.
|
||||
`logLine`, present only when `isLlamaSwap` is true, is the most recent
|
||||
REAL backend `llama-server` process log line (`load_model: ...`,
|
||||
`llama_server: model loaded`, etc.), sourced from the endpoint's own
|
||||
`GET /api/events` SSE stream and filtered to `source: "upstream"` frames
|
||||
only (never llama-swap's own `source: "proxy"` request-access log) — one
|
||||
connection is held open per endpoint and reused across every poller,
|
||||
idle-closed after 30s of nobody asking. This is what the Run-menu
|
||||
picker's loading banner polls once a second while a model is loading.
|
||||
- `POST /api/v1/sessions/:id/custom-model` with `{ endpointId, modelId,
|
||||
confirmed? } | { clear: true }` applies (or clears) the session's
|
||||
selection and **restarts the session's CLI process in place** — every
|
||||
supported harness reads its endpoint config at process start, never per
|
||||
turn, so there is no live hot-swap. (`POST /api/v1/quick-start`'s own
|
||||
`customModel: { endpointId, modelId, confirmed? }` field is the
|
||||
no-restart equivalent for a session that doesn't exist yet — see below.)
|
||||
A Claude session resumes its existing conversation across the restart;
|
||||
pi/omp/grok additionally get a forced `--model`/`-m` value, since for
|
||||
those three the config file alone does not select it. `400 INVALID_INPUT`
|
||||
for a remote (SSH) or Docker session — both restart their agent
|
||||
differently under the hood, and applying to one would report success
|
||||
while changing nothing. Two more responses replace the normal
|
||||
`{customModel, restarted}` shape, neither an error, and neither restarts
|
||||
or creates anything on the first ask. ⚠️ **Each is answered by its OWN
|
||||
flag on the retry, and answering one is not consent to the other**: they
|
||||
are questions about different people, and while they shared a single flag
|
||||
a caller who confirmed the context warning silently agreed to evict
|
||||
another session's model as well. Send `confirmedContext: true` to proceed
|
||||
past the context warning, `confirmedSwap: true` past the swap conflict,
|
||||
and both when both were asked (they accumulate, so the second retry still
|
||||
carries the first answer). The original `confirmed: true` still means
|
||||
BOTH and is still accepted, because it shipped in this feature's
|
||||
HTTP-API-only cut; new callers should send the specific one:
|
||||
- `{requiresConfirmation: true, currentlyLoadedModel, affectedSessions}` —
|
||||
llama.cpp/llama-swap only runs one model at a time, and switching would
|
||||
unload a model another **live session's own selection** is actively
|
||||
using. Never returned for a plain (non-llama-swap) server, and never
|
||||
just because a swap is needed at all — only when it would disrupt
|
||||
someone else.
|
||||
- `{requiresContextWarning: true, modelId, contextLength,
|
||||
minSafeContextTokens}` — Claude Code's own fixed per-turn overhead
|
||||
(system prompt + tool schemas) can exceed a small model's entire
|
||||
discovered context on its own, before any conversation history exists
|
||||
to compact, guaranteeing the very first message fails regardless of
|
||||
`CLAUDE_CODE_MAX_CONTEXT_TOKENS`. Gated on the CLI registry declaring a
|
||||
`contextLengthVar` (claude only today), so it never fires for another
|
||||
harness.
|
||||
- `POST /api/v1/quick-start`'s `customModel: { endpointId, modelId,
|
||||
confirmed?, confirmedContext?, confirmedSwap? }` field (alongside its
|
||||
normal `caseName`/`mode`/etc. body)
|
||||
computes the same injection **before** the session exists and launches
|
||||
directly on the endpoint — no restart, because there was never a
|
||||
native-backend boot to restart away from. Runs the identical checks as
|
||||
the dedicated route above (`requiresConfirmation`/`requiresContextWarning`,
|
||||
same shapes, same per-question `confirmedContext`/`confirmedSwap` retry),
|
||||
and is refused the same way
|
||||
for a remote or Docker case. This is what the Run-menu picker uses for
|
||||
opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP; Claude still uses the
|
||||
dedicated restart route above (its `--resume`-based restart is far less
|
||||
jarring than a full relaunch, and folding it into the one-shot path is
|
||||
separate work — see `docs/custom-model-endpoints-plan.md`).
|
||||
|
||||
## CLI management
|
||||
|
||||
Read and write the CLI registry (`docs/cli-registry.md`). Every **write** route answers `403 FORBIDDEN` while `cliManagementEnabled` is off (the default), and for a non-admin in multi-user mode. A write that would overwrite a `clis.json` which does not parse, or which has group/world permission bits, is refused with `409 CONFLICT` and a message naming the fix; the file is left untouched.
|
||||
|
||||
| Method | Path | Body | Notes |
|
||||
| -------- | ----------------------------- | ------------------------------------------------------- | ------------------------------------------------------------------------------------------------------- |
|
||||
| `GET` | `/api/clis` | none | Every entry, disabled ones included: `id`, `label`, `shortBadge`, `order`, `kind`, `enabled`, `stock`, `installed`, and `installCommand` for a stock entry. Not gated; a non-admin in multi-user mode gets `[]`. |
|
||||
| `PUT` | `/api/clis/:id` | `{ enabled }` | Toggle an existing entry, stock or custom. `404` for an unknown id; `400 INVALID_INPUT` when disabling a `kind: 'shell'` entry. |
|
||||
| `POST` | `/api/clis/:id/install` | none | Run a **stock** entry's install command (never a custom one: `400`). `409 CONFLICT` while an install for the same id is running; `422 OPERATION_FAILED` with the output tail when it fails. Never enables the entry. |
|
||||
| `POST` | `/api/clis` | `{ id, label, shortBadge, binaries, argv, enabled? }` | Create a custom entry. `409 ALREADY_EXISTS` for a stock id or an existing custom id. `enabled` defaults to `true`. |
|
||||
| `PUT` | `/api/clis/custom/:id` | `{ label, shortBadge, binaries, argv, enabled? }` | Replace an existing custom entry. An absent `enabled` keeps the entry's current state. `400` for a stock id, `404` for an unknown one. |
|
||||
| `DELETE` | `/api/clis/:id` | none | Delete a custom entry. `400` for a stock id, `404` for an unknown one. |
|
||||
|
||||
## Voice dictation
|
||||
|
||||
Browser dictation transcribed through this server's Claude Code login, i.e. the
|
||||
|
||||
File diff suppressed because one or more lines are too long
@@ -0,0 +1,314 @@
|
||||
# CLI management Settings UI + write API — plan
|
||||
|
||||
> Tracked separately from `DEPLOYMENT_PLAN.md` (PR B2, merged) and `docs/copilot-integration-plan.md`
|
||||
> (parked). This is "PR C" from the original #343 review: *"settings UI + write endpoints +
|
||||
> auto-install, once we've settled the trust model... I want to make that call on its own, not
|
||||
> inside a 100-file diff."*
|
||||
>
|
||||
> **Phase 0 is CLOSED as of 2026-09-21** — all three original pieces are IN SCOPE (expanded from
|
||||
> this plan's first draft, which recommended #2/#3 as separate/out-of-scope; the user chose full
|
||||
> scope instead, with the risk called out explicitly for #3 before confirming). See "Decisions"
|
||||
> below for the full record.
|
||||
|
||||
## Status as of 2026-09-22
|
||||
|
||||
**Phases 1–6 are ALL IMPLEMENTED** (commits `da07b38c` "add cliManagementEnabled flag and GET
|
||||
/api/clis" and `db4557d9` "Phases 3-6 - write API + custom entries + Settings UI", both on this
|
||||
branch, `feat/cli-management`). Confirmed present in the tree: `cliManagementEnabled` in
|
||||
`SettingsUpdateSchema`; `GET /api/clis`, `PUT /api/clis/:id`, `POST /api/clis/:id/install`,
|
||||
`POST /api/clis`, `PUT /api/clis/custom/:id`, `DELETE /api/clis/:id` in
|
||||
`src/web/routes/cli-registry-routes.ts`; the `shell`/`claude` `UNDISABLEABLE_IDS` backend guard;
|
||||
`isAdmin(req)` gating on both the list and write routes; `appendAdminAudit` wired into the install
|
||||
route; tmp+rename+`0o600` writes in `registry-writer.ts`; the full Settings UI (row list, toggle,
|
||||
Install button, custom-entry create/edit/delete form) in `settings-ui.js` + `index.html`.
|
||||
`test/routes/cli-registry-routes.test.ts` (425 lines) and `test/cli-registry-no-id-branching.test.ts`
|
||||
cover it. This status section, plus the fix and gap below, is the one piece of that work done in
|
||||
a *different* session from the one that wrote Phases 1–6 — reviewed by reading the diff and
|
||||
verifying each claim against the actual routes/tests, not by re-implementing anything.
|
||||
|
||||
### Gotcha found and fixed (commit `0c77dd0a`)
|
||||
|
||||
**Toggling a CLI off in Settings had no effect anywhere except the Settings row itself.**
|
||||
`window.__codemanCliAvailable` — the flag `isCliAvailable()` reads client-side to gate the
|
||||
welcome-screen buttons, the Run-menu dropdown and the mobile overview — is injected **once**, at
|
||||
initial page render (`server.ts`), built purely from each CLI's own installed-on-PATH resolver
|
||||
(`isClaudeAvailable()` etc.), with **no reference to the registry's `enabled` flag at all**. So
|
||||
disabling a CLI here updated its own row and nothing else — every launch surface kept offering it,
|
||||
both live and after a full page reload, since even a *fresh* render never consulted the registry.
|
||||
Root-caused and reported by the user testing the live feature ("toggle those off, they still
|
||||
appear in that menu and on the front main screen").
|
||||
|
||||
Fixed two places:
|
||||
- `server.ts`: after building `available`, intersect the nine real `SessionMode` ids against
|
||||
`enabledClis()`. `git`/`cloudflared` (utility binaries, not CLI registry entries) and
|
||||
`deepseekBinary` (a secondary installed-only flag for the "add a profile" affordance) are
|
||||
deliberately left alone — they were never registry-gated to begin with.
|
||||
- `settings-ui.js`: `toggleCliEnabled()` now patches `window.__codemanCliAvailable` in place and
|
||||
refreshes the welcome screen, the mobile overview and an already-open Run menu, mirroring the
|
||||
existing `installDeepSeekProfile()` pattern for the same "injected once, needs an explicit
|
||||
patch" reason — the server-side fix alone still left every surface stale until the next reload.
|
||||
|
||||
New test in `test/render-index-html.test.ts`: an installed-but-disabled CLI (codex, forced via
|
||||
`clis.json` + `reloadCliRegistry()`) reads as unavailable, while an installed-and-enabled one
|
||||
(claude) is unaffected by the override.
|
||||
|
||||
**Verified on the Debian devbox** (`codeman-devbox`, real tmux — this sandbox has none and
|
||||
`WebServer`'s constructor hard-requires it): typecheck clean, the new test passes (17/17 in
|
||||
`render-index-html.test.ts`), the CLI-registry suites pass (86/86), and the **full CI gate is
|
||||
green — 415 test files, 7855 tests, 0 failures**.
|
||||
|
||||
### Launch-surface registry integration — completed
|
||||
|
||||
The welcome screen, desktop Run menu and mobile Run picker now use the same injected CLI catalog.
|
||||
Every enabled registry entry is rendered; unavailable binaries remain hidden as before. Settings
|
||||
updates the catalog and availability flags in place after enable/disable, create, edit or delete,
|
||||
so the launch surfaces update without a page reload. A custom entry uses the generic quick-start
|
||||
path, while stock entries retain their existing per-CLI launch settings.
|
||||
|
||||
Not otherwise re-verified line-by-line against every Phase 1–6 checklist item below (e.g. the
|
||||
exact wording of toasts, the "same PR" sequencing notes) — the checklists are left as originally
|
||||
written; treat the **Status** section above as authoritative for what exists.
|
||||
|
||||
---
|
||||
|
||||
## Background
|
||||
|
||||
`src/config/cli-registry/registry.ts` is READ-ONLY today, and says so in its own header comment:
|
||||
|
||||
> "⚠️ READ-ONLY. Nothing in this module writes, creates or migrates the file... there is no
|
||||
> settings UI and no write API yet... A `seededStockIds` ratchet belongs with the write API that
|
||||
> needs it."
|
||||
|
||||
Confirmed on `master` (2026-09-21): no `/api/clis` route exists at all (read or write);
|
||||
`~/.codeman/clis.json` is hand-edit-only; `resolveInstallCommandForPlatform()` is documented
|
||||
"Display text only — never executed" — nothing runs an install command server-side today. The
|
||||
original #343 review flagged the opposite (`spawn(command, {shell: true})`, `env.allowedPrefixes`
|
||||
contributed from a write) as needing its own trust-model decision; that decision was never made
|
||||
after the split, just dropped. This plan makes it.
|
||||
|
||||
**Closest existing precedent, and the template this plan follows for the read/write API**:
|
||||
`src/web/routes/custom-model-routes.ts` + `src/custom-model-hosts.ts` (#393/#430/#459) — a small
|
||||
per-item JSON store, Settings-UI-driven, admin-gated in multi-user mode, tmp+rename+0600 writes.
|
||||
|
||||
**Precedent for the new master feature flag (Phase 1)**: `customModelEndpointsEnabled` —
|
||||
`z.boolean().optional()` in `SettingsUpdateSchema` (`schemas.ts:1319`), a checkbox read/written by
|
||||
id in `openAppSettings()`/`saveAppSettings()` (`settings-ui.js:401`/`:2120`). SYNCED, not
|
||||
per-device (present in the schema, absent from `displayKeys`), default OFF.
|
||||
|
||||
**Spec refs for the whole plan:**
|
||||
- `src/config/cli-registry/registry.ts` — the read path; `resolveRegistry()`'s merge semantics
|
||||
(`deepMerge`, `UNMERGEABLE_KEYS`) apply unchanged to whatever this plan writes
|
||||
- `docs/cli-registry.md` — registry shape, "The override file", "Arg-template safety" (the four
|
||||
layers Phase 5's custom-entry validation must not weaken), "Adding a CLI" (the 5-step recipe a
|
||||
custom entry does NOT get to skip just because it arrives via UI instead of a stock.ts edit)
|
||||
- `src/web/routes/custom-model-routes.ts` + `src/custom-model-hosts.ts` — read/write API template
|
||||
- `docs/multi-user-plan.md`, `docs/security-architecture.md` — admin-gating conventions
|
||||
- `CLAUDE.md` §Multi-user mode, §"Settings surface", §"Per-device vs synced settings"
|
||||
|
||||
---
|
||||
|
||||
## Decisions (Phase 0, closed 2026-09-21)
|
||||
|
||||
1. **Enable/disable a stock CLI's `enabled` flag** — IN SCOPE. Plus a **master feature flag**
|
||||
(`cliManagementEnabled`, synced, default OFF) gating the whole Settings UI section's visibility,
|
||||
matching this codebase's standing convention for new admin-facing surfaces.
|
||||
2. **Auto-install** (stock CLIs' already-shipped, already-vetted install commands) — IN SCOPE,
|
||||
same PR.
|
||||
3. **Custom CLI entries via the UI** — IN SCOPE, **typed-argv only**: a custom entry goes through
|
||||
the exact same schema/argv-safety path stock entries do (named token patterns, no raw shell-text
|
||||
field). Its install command stays **display-only text**, same as every stock entry today — Phase
|
||||
4's auto-install NEVER executes a custom entry's install command, only a stock one's. This is
|
||||
the one place scope was deliberately narrowed relative to what was agreed in principle, because
|
||||
`docs/cli-registry.md`'s arg-template-safety section exists specifically to keep config free of
|
||||
shell text, and a free-text install command for a user-defined entry would reopen exactly that.
|
||||
4. **`shell`/`claude` un-disableable** — enforced at the **backend**, not just the UI (a
|
||||
frontend-only guard is bypassable with curl).
|
||||
5. **Non-admin visibility in multi-user mode** — the CLI-management Settings section is **hidden
|
||||
entirely** for a non-admin, not shown-empty.
|
||||
6. **`seededStockIds` ratchet** — not needed. `deepMerge()` only overrides a key the file actually
|
||||
sets, so a CLI absent from `clis.json.clis` always falls through to its stock `enabled` value
|
||||
with no special-casing. (Carried over from the first draft, not re-litigated.)
|
||||
|
||||
---
|
||||
|
||||
## Phase 1 — Master feature flag: `cliManagementEnabled`
|
||||
|
||||
**Status:** DONE (commit `da07b38c`) — verified present in `SettingsUpdateSchema`, `index.html`,
|
||||
`openAppSettings()`/`saveAppSettings()`.
|
||||
|
||||
**Spec refs:**
|
||||
- `schemas.ts:1319` (`customModelEndpointsEnabled`) — the exact pattern to mirror: `z.boolean().optional()`
|
||||
in `SettingsUpdateSchema`
|
||||
- `settings-ui.js:401`/`:2120` — checkbox read/write by id in `openAppSettings()`/`saveAppSettings()`
|
||||
- `CLAUDE.md` §"Adding Features" → "App setting" — decide per-device vs synced FIRST (this one is
|
||||
synced: a feature toggle, not a display preference) and add to `displayKeys` NEVER for a synced
|
||||
setting
|
||||
|
||||
**Checklist:**
|
||||
- [x] Add `cliManagementEnabled: z.boolean().optional()` to `SettingsUpdateSchema`
|
||||
- [x] Add the checkbox to `index.html`'s `#settings-clis` section, above where Phase 6's per-CLI
|
||||
list will render — reads/writes via `openAppSettings()`/`saveAppSettings()` by id, same as
|
||||
`customModelEndpointsEnabled`
|
||||
- [x] `readCliManagementEnabled()` helper (mirrors `readCustomModelEndpointsEnabled()` in
|
||||
`custom-model-routes.ts:609`) for the route file(s) in Phases 2-5 to gate on
|
||||
- [x] When OFF: `GET /api/clis` still exists but the Settings UI section stays hidden
|
||||
(`applyCliManagementVisibility()`); the write endpoints reject (see Phase 3)
|
||||
|
||||
**Verify:** `npm run typecheck` passes; a unit test confirms `SettingsUpdateSchema` accepts/rejects
|
||||
the field correctly; toggling it in a fresh browser profile shows/hides the Settings section with
|
||||
no server restart.
|
||||
|
||||
---
|
||||
|
||||
## Phase 2 — Read endpoint: `GET /api/clis`
|
||||
|
||||
**Status:** DONE (commit `da07b38c`) — verified present in `src/web/routes/cli-registry-routes.ts`.
|
||||
|
||||
**Spec refs:**
|
||||
- `src/web/routes/custom-model-routes.ts:730` (`GET /api/model-endpoints`) — multi-user read
|
||||
gating: empty list for a non-admin, never a 403
|
||||
- `src/config/cli-registry/registry.ts` — `listClis()` (every entry, including disabled stock
|
||||
ones — this is an admin/settings surface, unlike `enabledClis()`)
|
||||
- `window.__codemanCliAvailable`'s resolvers (`isClaudeAvailable()` etc.) — candidate `installed`
|
||||
source; confirm whether to reuse directly or the response needs its own probe (Open Question 4,
|
||||
carried from the first draft — still genuinely open, decide during this phase not before)
|
||||
|
||||
**Checklist:**
|
||||
- [x] New route file `cli-registry-routes.ts`
|
||||
- [x] Response excludes `launch`/`env`/`capabilities`/`overlays`/`discovery`
|
||||
- [x] `isMultiUserMode() && !isAdmin(req)` → `[]`
|
||||
- [x] Unit tests in `test/routes/cli-registry-routes.test.ts` (admin/non-admin/single-user,
|
||||
disabled stock CLI still present)
|
||||
|
||||
**Verify:** `npm test -- test/routes/cli-registry-routes.test.ts` passes; `curl localhost:3000/api/clis | jq`
|
||||
shows every stock CLI including disabled ones.
|
||||
|
||||
---
|
||||
|
||||
## Phase 3 — Write endpoint: `PUT /api/clis/:id` (stock enable/disable)
|
||||
|
||||
**Status:** DONE (commit `db4557d9`) — `UNDISABLEABLE_IDS`, admin gate, tmp+rename+0600 all
|
||||
confirmed present.
|
||||
|
||||
**Spec refs:**
|
||||
- `src/web/routes/custom-model-routes.ts:753` + `src/custom-model-hosts.ts:91` — write-path
|
||||
template: `adminOnly` gate, read-modify-write the WHOLE file, tmp+rename+0600
|
||||
- `registry.ts:47` (`filePath()` = `dataPath(...)`) and `reloadCliRegistry()` — write to the same
|
||||
resolved path, invalidate the cache on every successful write or the change is invisible until
|
||||
restart
|
||||
|
||||
**Checklist:**
|
||||
- [x] Body: `{ enabled: boolean }`. Zod schema in `schemas.ts`
|
||||
- [x] Gate order: `cliManagementEnabled` → `adminOnly` → shell/claude guard → stock-only guard
|
||||
- [x] Rejects disabling `shell` or `claude` (`UNDISABLEABLE_IDS`)
|
||||
- [x] Rejects a write for an id that isn't a stock CLI
|
||||
- [x] Deep-merges `{ clis: { [id]: { enabled } } }`, preserving other override keys
|
||||
- [x] tmp+rename+0600 write, `reloadCliRegistry()` on success
|
||||
- [x] Unit tests (`test/routes/cli-registry-routes.test.ts`)
|
||||
|
||||
**Verify:** `npm test` full gate green; `curl -X PUT localhost:3000/api/clis/grok -d '{"enabled":false}'`
|
||||
then `GET /api/clis` shows the change with no restart; same against `shell`/`claude` returns an
|
||||
error and changes nothing; `ls -la ~/.codeman/clis.json` shows mode 0600.
|
||||
|
||||
---
|
||||
|
||||
## Phase 4 — Auto-install: `POST /api/clis/:id/install` (stock CLIs only)
|
||||
|
||||
**Status:** DONE (commit `db4557d9`) — route present, `appendAdminAudit` wired in.
|
||||
|
||||
**Spec refs:**
|
||||
- `registry.ts:231` (`resolveInstallCommandForPlatform`) — currently "Display text only — never
|
||||
executed"; this phase is what changes that, for stock entries only, with Decision 2's sign-off
|
||||
- Original #343 review's exact concern re: `env.allowedPrefixes` contributed from a write — stays
|
||||
out of scope; this phase only ever runs a command, never touches the env allowlist
|
||||
|
||||
**Checklist:**
|
||||
- [x] Separate endpoint from Phase 3's toggle
|
||||
- [x] Gate order: `cliManagementEnabled` → `adminOnly` → stock-entry-only guard
|
||||
- [x] `resolveInstallCommandForPlatform(entry)` for the target
|
||||
- [x] Bounded execution (timeout, captured stdout/stderr)
|
||||
- [x] Does NOT auto-enable on successful install
|
||||
- [x] Audit-logged via `appendAdminAudit`
|
||||
- [x] Unit tests
|
||||
|
||||
**Verify:** a real install triggered via the endpoint against a CLI not currently installed,
|
||||
`GET /api/clis`'s `installed` field flips true with no restart; audit log entry present; attempting
|
||||
install against a custom entry's id fails with a clear error; full CI gate green.
|
||||
|
||||
---
|
||||
|
||||
## Phase 5 — Custom CLI entries: create / update / delete via API
|
||||
|
||||
**Status:** DONE (commit `db4557d9`) — `POST /api/clis`, `PUT /api/clis/custom/:id`,
|
||||
`DELETE /api/clis/:id` all present. Open Question 2 resolved: a **separate** endpoint
|
||||
(`PUT /api/clis/custom/:id`), not Phase 3's `PUT /api/clis/:id` widened.
|
||||
|
||||
**Spec refs:**
|
||||
- `docs/cli-registry.md` §"Arg-template safety" (all four layers), §"Adding a CLI" (the 5-step
|
||||
recipe) — a custom entry created via this API must satisfy the SAME schema (`CliEntrySchema`)
|
||||
every stock entry does; there is no relaxed path for UI-originated entries
|
||||
- `registry.ts`'s `resolveRegistry()` — the custom-entry branch (`stock: false`, dropped with a
|
||||
warning on validation failure, never falls back silently) already exists and is unchanged by
|
||||
this phase; this phase only adds a way to WRITE what that branch reads
|
||||
|
||||
**Checklist:**
|
||||
- [x] `POST /api/clis` (create), full `CliEntrySchema` validation
|
||||
- [x] `PUT /api/clis/custom/:id` (update) — separate endpoint from Phase 3's stock toggle
|
||||
- [x] `DELETE /api/clis/:id` refuses for any stock id
|
||||
- [x] `id` collision check against existing stock ids
|
||||
- [x] `discovery.install.command` on a custom entry stays DISPLAY-ONLY
|
||||
- [x] Same tmp+rename+0600 write pattern, `reloadCliRegistry()` on every successful mutation
|
||||
- [x] Unit tests
|
||||
|
||||
**Verify:** `npm test` full gate green; create a custom entry via curl, confirm it appears in
|
||||
`GET /api/clis` — **confirm it appears in the Run menu is UNVERIFIED and currently FALSE, see
|
||||
"Outstanding" above**; delete it, confirm it's gone and `clis.json` no longer references it.
|
||||
|
||||
---
|
||||
|
||||
## Phase 6 — Settings UI
|
||||
|
||||
**Status:** DONE (commit `db4557d9`) — `#cliListGroup`, row rendering, toggle, Install button,
|
||||
custom-entry create/edit/delete form all present in `settings-ui.js`/`index.html`. Manual browser
|
||||
verification per the phase's own "Verify" step (flag on/off, non-admin hidden, toggle stops the
|
||||
Run menu offering a CLI, create/enable/launch a custom entry, delete it, shell/claude undisableable)
|
||||
has **not** been re-run in this session — the toggle→Run-menu leg specifically was BROKEN until the
|
||||
gotcha fix above, and the create→launch leg for a custom entry is the confirmed gap in
|
||||
"Outstanding".
|
||||
|
||||
**Spec refs:**
|
||||
- `index.html:2357` (`#settings-clis`) — the existing home; Phase 1's master toggle at the top,
|
||||
then the per-CLI list, then (if `cliManagementEnabled`) a "custom CLI" creation form, all above
|
||||
the existing Codex-only groups
|
||||
- `CLAUDE.md` §"Settings surface" — App Settings scrolls, it does not tab-switch
|
||||
- `admin-ui.js` — pattern for an admin-only-VISIBLE section (not just admin-only-writable),
|
||||
needed here per Decision 5
|
||||
|
||||
**Checklist:**
|
||||
- [x] Whole section hidden when `cliManagementEnabled` is OFF, and separately hidden for a
|
||||
non-admin in multi-user mode (`_applyCliManagementAdminGate`)
|
||||
- [x] Fetches `GET /api/clis` when the section becomes visible; renders one row per CLI
|
||||
- [x] Stock rows: enabled toggle only; `shell`/`claude` rows show the toggle disabled/greyed
|
||||
- [x] Custom rows: enabled toggle plus edit/delete affordances
|
||||
- [x] "Add custom CLI" form (id/label/badge/binary/argv)
|
||||
- [x] Toggle/edit/delete update the row in place
|
||||
|
||||
**Verify:** manual browser test per `CLAUDE.md`'s "Always Test Before Deploying" rule — **not yet
|
||||
re-run end-to-end in this session**; do this before considering the feature ready to ship, and
|
||||
expect the custom-entry-launch step to fail until the Outstanding gap above is closed.
|
||||
|
||||
---
|
||||
|
||||
## Remaining Open Questions
|
||||
|
||||
1. **Phase 2's `installed` source** — resolved: reuses `window.__codemanCliAvailable`'s existing
|
||||
resolvers via `GET /api/clis`'s own probe (confirmed by reading the route).
|
||||
2. **Phase 5's `PUT` endpoint shape** — resolved: a **separate** endpoint
|
||||
(`PUT /api/clis/custom/:id`), not Phase 3's toggle route widened.
|
||||
3. **Sequencing against the parked Copilot plan** — unchanged, still not blocking.
|
||||
4. **NEW: custom-CLI Run-menu integration** — see "Outstanding" above. Not decided or started.
|
||||
|
||||
---
|
||||
|
||||
Implementation is underway (see Status above); this line is left for history rather than removed —
|
||||
the plan was originally approved before Phases 1–6 landed.
|
||||
+70
-5
@@ -18,7 +18,17 @@ Every run mode Codeman can launch — Claude Code, Terminal/Shell, OpenCode, Cod
|
||||
|
||||
## The override file
|
||||
|
||||
`~/.codeman/clis.json` (instance-scoped through `dataPath()`) holds overrides and custom entries only, never a copy of the stock catalog: `{ "clis": { "<id>": { ...partial entry... } } }`. Objects merge key-wise onto the stock entry, arrays replace wholesale. **The file must be mode 0600**; the loader refuses any group/world permission bit, read bits included, so a file created with a normal umask (0644) is ignored until you `chmod 600` it. Every reason a file was ignored or an entry dropped is logged once, prefixed `[cli-registry]`, on the first load. A stock entry whose override fails validation falls back to the shipped definition; a custom entry that fails is dropped. The file is read once per process and re-read only on restart.
|
||||
`~/.codeman/clis.json` (instance-scoped through `dataPath()`) holds overrides and custom entries only, never a copy of the stock catalog: `{ "clis": { "<id>": { ...partial entry... } } }`. Objects merge key-wise onto the stock entry, arrays replace wholesale. **The file must be mode 0600**; the loader refuses any group/world permission bit, read bits included, so a file created with a normal umask (0644) is ignored until you `chmod 600` it. Every reason a file was ignored or an entry dropped is logged once, prefixed `[cli-registry]`, on the first load. A stock entry whose override fails validation falls back to the shipped definition; a custom entry that fails is dropped. The file is read once per process and re-read after a change made through CLI management (below).
|
||||
|
||||
## Managing CLIs from Settings
|
||||
|
||||
App Settings → Agents & CLIs → **CLI management** (`cliManagementEnabled`, default OFF; admin-only in multi-user mode) lists every entry with an installed/not-installed badge and:
|
||||
|
||||
- toggles any entry on or off. A `kind: 'shell'` entry cannot be disabled, and the row shows no switch for it. A disabled CLI disappears from the Run menu, the welcome screen and the phone overview, and new session requests for it are rejected.
|
||||
- installs a missing **stock** CLI by running its shipped install command, after a confirm that names the exact command. Only one install per CLI runs at a time, and the command runs without any `CODEMAN_*` variable in its environment. A custom entry's install command is never executed.
|
||||
- adds, edits and deletes **custom** entries (id, label, badge, binaries, launch argv). The server re-validates the whole assembled entry through `CliEntrySchema`, so the form cannot bypass the load-time rules.
|
||||
|
||||
These are the only writes to `clis.json`. They are serialized, and a file that does not parse or has unsafe permissions is refused rather than overwritten; fix it (or `chmod 600` it) and retry. The HTTP routes are listed in `docs/api-reference.md` under *CLI management*.
|
||||
|
||||
## The shape of an entry
|
||||
|
||||
@@ -36,7 +46,9 @@ interface CliEntry {
|
||||
launch: CliLaunch; // the structured argv template
|
||||
env: CliEnv; // exports, tmux setenv keys, the env-override allowlist
|
||||
capabilities: CliCapabilities; // what every call site reads instead of the id
|
||||
// .workDetect?: { promptGlyph, workingLine } — how this CLI's pane shows work
|
||||
// .workDetect?: { promptGlyph, workingLine, watchingLine?, watchingLines?, awaitingLine? }
|
||||
// — how this CLI's pane shows work, work it started in the background, and a turn
|
||||
// that ended waiting for workers it will resume from
|
||||
overlays: CliOverlays; // remote-SSH / Docker pane commands, credential store
|
||||
}
|
||||
```
|
||||
@@ -45,10 +57,61 @@ interface CliEntry {
|
||||
|
||||
### Regexes that come from config
|
||||
|
||||
Two capability fields carry a regular expression an override file can set: `discovery.version.regex` and `capabilities.workDetect.workingLine`. Both go through `compileVersionRegex()`, which caps the source at 200 characters, refuses the nested-quantifier shapes that cause catastrophic backtracking, and returns `null` rather than throwing so every caller degrades instead of crashing.
|
||||
Four capability fields carry a regular expression an override file can set: `discovery.version.regex`, `capabilities.workDetect.workingLine`, `capabilities.workDetect.watchingLine` and `capabilities.workDetect.awaitingLine`. All four go through `compileVersionRegex()`, which caps the source at 200 characters, refuses the nested-quantifier shapes that cause catastrophic backtracking, and returns `null` rather than throwing so every caller degrades instead of crashing.
|
||||
|
||||
`workingLine` is the one that matters most, because it is compiled once per session and then run against every accumulated PTY chunk and every pane capture. A nested quantifier there is a ReDoS against the event loop for the whole server, not just that session. The guard therefore runs in two places, and neither is redundant: `schema.ts` rejects the entry at LOAD time so a bad pattern never reaches a session, and `_workingLinePattern()` in `session.ts` compiles through the same helper so the runtime cannot end up with a pattern the schema would have refused.
|
||||
|
||||
`watchingLine` reads a different row of the same screen. A CLI draws it while work the agent
|
||||
itself started is still running — Claude prints `⏵⏵ bypass permissions on · 1 monitor · ← for
|
||||
agents` while a monitor, a backgrounded shell or a cloud session is live. Codeman turns that
|
||||
into `Session.watching`, and an idle prompt from such a session opens already acknowledged,
|
||||
so a pane waiting for its own background work never raises an alert a human cannot answer.
|
||||
Group 1 is the label, and a CLI that declares no pattern reports no background work.
|
||||
Claude's Artifact comment monitor is the one chip that does not count. It waits for a human
|
||||
to comment on a page the agent published, so Claude's pattern refuses any footer that
|
||||
carries it, and the idle alert goes out as usual.
|
||||
|
||||
Two CLIs declare such a row today, and they put it in different places. Claude writes its
|
||||
chip on the last row of the screen, so it keeps the default one-row window and anchors on
|
||||
the `·` its footer joins items with. Codex pins
|
||||
`1 background terminal running · /ps to view · /stop to close` ABOVE its composer, which
|
||||
puts the row third from the bottom once the status line and the composer are counted, so its
|
||||
entry declares `watchingLines: 3` and matches that row end to end. Both were measured
|
||||
against live panes rather than read out of a binary, which is the standard for adding a
|
||||
third.
|
||||
|
||||
`awaitingLine` covers the quiet pane that is neither idle nor watching: a turn that ENDED
|
||||
to wait for workers the CLI will resume from by itself. When background agents or an
|
||||
ultracode workflow are still running at turn end, Claude closes the turn with
|
||||
`✻ Waiting for 1 dynamic workflow to finish` instead of `✻ Brewed for 1m 18s`, and a pane
|
||||
showing that row counts as working. ⚠️ Claude renders the row once and never redraws it, so
|
||||
the words are still on screen after the workers report back and the follow-up turn ends.
|
||||
The pattern is therefore never run over the whole pane: `isAwaitingWorkers()`
|
||||
(`session-activity.ts`) walks up from the composer past blank, framed and indented rows and
|
||||
tests only the first row that starts in column 0, which is the newest transcript row. Claude
|
||||
starts its own rows in column 0 and the agent's prose never does, so the anchor also keeps an
|
||||
agent from holding its own session busy.
|
||||
|
||||
That label is the one value in the registry that an AGENT can influence, because it comes off
|
||||
the agent's own screen. Two things keep it honest, and both belong to whoever adds a pattern
|
||||
for a new CLI. `watchingLabel()` in `session-activity.ts` searches only the last few
|
||||
non-blank rows, which should be the part of the screen the CLI draws rather than the agent,
|
||||
and the pattern should anchor on chrome only that CLI can produce. Keep the window as small
|
||||
as the layout allows, since every row it adds is another row the agent may be able to write.
|
||||
The label is also ANSI-stripped and length-capped at the source, and every interpolation of
|
||||
it into markup goes through `escapeHtml()`, since it ends up on a badge and in an approval
|
||||
card.
|
||||
|
||||
The two shipped entries do not sit equally well behind that rule, and the difference decides
|
||||
what a pattern is allowed to do. Claude's chip is the last row, so its one-row window holds
|
||||
nothing the agent can write — not even the status line above it, whose command a session
|
||||
running with permissions bypassed can write into its own `.claude/settings.json`. Codex's row
|
||||
shares its slot with the last row of the transcript whenever no terminal is running, so a
|
||||
message ending in that exact line is matched. What keeps that harmless is `hooks: 'none'`: no
|
||||
hook event from a codex session reaches the approvals inbox, so a forged label costs a wrong
|
||||
badge and cannot silence an alert. Before giving a CLI both hook signals and a pattern, make
|
||||
sure its row is one the agent cannot write.
|
||||
|
||||
### Three capabilities that must stay independent
|
||||
|
||||
`external`, `hooks` and `altScreen` describe three different, deliberately unequal sets, and deriving any one from another has already shipped a bug. `shell` has no hooks but is **not** an external CLI, so a hooks predicate written as `!isExternalCliMode()` accepted `until=stop` on a shell session and then blocked the caller for their entire timeout. `deepseek` is the mirror image: it IS external and it DOES have hooks.
|
||||
@@ -102,6 +165,8 @@ It matches four shapes, not one: `mode === '<id>'`, `mode !== '<id>'`, `case '<i
|
||||
|
||||
The allowlist is not a formality. If a branch is about what a CLI can DO it belongs in `CliCapabilities`; the entries that remain are things that are not CLI-behaviour branches at all — chiefly the legacy per-mode `<Mode>Config` objects on `POST /api/sessions`, which are a fact about the public HTTP API rather than about any CLI, plus a few documented cases where `mode === 'claude'` is genuinely the right question (Read My Mind reads Claude's _own_ transcript, so a capability there would be actively wrong).
|
||||
|
||||
`test/frontend-cli-no-id-branching.test.ts` is the same guard for the two frontend files the CLI registry's Run-menu consolidation touches, `session-ui.js` and `mobile-overview.js` — deliberately not the rest of `src/web/public/`, whose per-CLI rules stay out of scope for now (see "Fields declared for later" below). Its allowlist keys on `<file>::<expression>` with no line number, since a single unrelated edit to a contended file would otherwise shift every subsequent line and make every entry go stale at once, and each entry additionally carries the exact number of approved call sites — a bare key would let a brand-new branch reusing an already-approved expression land unreviewed. Its comparison shape differs from the backend guard's in one respect: the left-hand side may be any identifier, not only one named `mode`, `id` or `agentType`, because the review of #458 found `const m = this._runMode; if (m === 'codex')` slipping past the named form while the scanned file already filters with `(m) => m !== 'shell'`.
|
||||
|
||||
## Two namespaces called `param`
|
||||
|
||||
`launch.params` keys, `env.configSetenv[].fromParam` and `capabilities.privilegedParams[].param` all name a **launch param**. The **legacy wire field** a param arrives as is a separate namespace, and `launch.legacyConfigAliases` is the only bridge between the two.
|
||||
@@ -110,9 +175,9 @@ This matters because it is invisible when it is wrong. `capabilities.privilegedP
|
||||
|
||||
## Fields declared for later
|
||||
|
||||
`shortBadge`, `accent`, `capabilities.echo`, `capabilities.wheelForward`, `capabilities.keyboardAccessory` and `capabilities.maxFrameBytes` are **declared but not yet read**. They all describe frontend behaviour, and the frontend is deliberately untouched here: `app.js`, `terminal-ui.js` and `styles.css` keep their own hand-authored per-CLI rules, and moving them is its own piece of work verified by a browser/mobile suite the CI gate cannot see.
|
||||
`accent`, `capabilities.echo`, `capabilities.wheelForward`, `capabilities.keyboardAccessory` and `capabilities.maxFrameBytes` are **declared but not yet read**. (`shortBadge` was on this list until the CLI management list in Settings started showing it.) They all describe frontend behaviour, and the frontend is deliberately untouched here: `app.js`, `terminal-ui.js` and `styles.css` keep their own hand-authored per-CLI rules, and moving them is its own piece of work verified by a browser/mobile suite the CI gate cannot see.
|
||||
|
||||
Treat those values as **transcribed, not authoritative** — nothing enforces that `echo.policy` matches `_updateLocalEchoState`'s fallthrough, or that `accent` matches the gradient CSS paints, so re-measure before wiring one up. A field that is both wrong and unread is worse than an absent one, because the next reader trusts it; `test/cli-registry-no-id-branching.test.ts` pins the list so it cannot quietly grow, and wiring one up makes its line there fail, which is the direction you want.
|
||||
Treat those values as **transcribed, not authoritative** — nothing enforces that `echo.policy` matches `_updateLocalEchoState`'s fallthrough, so re-measure before wiring one up. `accent` is the one exception: it was measured against styles.css on 2026-09-21 (method in the comment above `CLAUDE` in `stock.ts`), though nothing keeps it in step with the CSS either. A field that is both wrong and unread is worse than an absent one, because the next reader trusts it; `test/cli-registry-no-id-branching.test.ts` pins the list so it cannot quietly grow, and wiring one up makes its line there fail, which is the direction you want.
|
||||
|
||||
`overlays.credStore` is in the same category, for a sharper reason: the Docker credential-seeding path still reads its own `CRED_STORES` table, because this shape allows ONE store per CLI and the live table needs two for gemini (`.gemini` for the CLI's own auth plus `.config/gcloud` for Vertex), while deepseek declares none here even though `.dsh` is seeded. Wiring it means making the field an array and correcting those two entries — a change to credential seeding, which is simultaneously the worst thing here to get wrong and the least covered by tests, since every docker IO path is no-op'd under vitest.
|
||||
|
||||
|
||||
@@ -104,17 +104,17 @@ declared capability, never an `if (mode === 'claude')` branch.
|
||||
|
||||
## Per-CLI injection recipes (confidence-ranked)
|
||||
|
||||
| CLI | Mechanism | Confidence |
|
||||
| ------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `claude` | Env vars: `ANTHROPIC_BASE_URL`, `ANTHROPIC_API_KEY`, `ANTHROPIC_DEFAULT_SONNET_MODEL`/`_HAIKU_MODEL`/`_OPUS_MODEL` (all set to the chosen model/deployment name) | **Verified end-to-end** against a real llama-swap server — a real "hello world" reply came back. ⚠️ Non-interactive (`-p`) invocations also fire an async session-title-generation call that reuses `ANTHROPIC_DEFAULT_HAIKU_MODEL` and validates it against Claude Code's OWN internal recognized-model list, printing `[claude-code:unrecognized_model]` and, in `-p` mode, hanging the whole invocation rather than just warning. `--settings '{"autoTitle":false}'` does NOT stop this (confirmed); `--bare` does (the warning still prints, but the real prompt runs) — but `--bare` ALSO disables hooks, LSP, plugin sync, and CLAUDE.md auto-discovery, so it is only safe for the standalone one-shot test script, NEVER for a real interactive Codeman session (which depends on hooks for idle detection, trust-dialog auto-accept, etc. — see the External CLI modes section of CLAUDE.md). Whether an INTERACTIVE claude session with a custom model hits the same hang (vs. just a background warning) is untested and should be checked before calling chunk 5/6 done for claude |
|
||||
| `opencode` | `OPENCODE_CONFIG_CONTENT` env var (already a registry mechanism, `stock.ts:342`) holding a JSON blob: `{"provider":{"custom":{"options":{"baseURL":...,"apiKey":...},"models":{"<name>":{}}}},"model":"custom/<name>"}` | **Verified by user** |
|
||||
| `codex` | TOML `config.toml`: top-level `model = "<id>"` + `[model_providers.custom]` (`base_url`, `env_key` naming an env var the real API key rides in — never a literal TOML field, since codex's schema has no such field). Written to an isolated dir via `CODEX_HOME` (`stock.ts:405-415`) so the user's own `~/.codex/config.toml` is never touched | **Config STRUCTURE verified** against a real codex binary (an earlier `[model].default` table shape was rejected: "invalid type: map, expected a string" — caught live). **Protocol CONFIRMED BROKEN against llama.cpp/llama-swap**: codex only speaks the Responses API (`wire_api = "responses"`, the only value it accepts since it dropped `"chat"` support in Feb 2026), and a real llama-swap server does not implement `/v1/responses` — a live run against it failed with repeated `Reconnecting...` then `high demand` errors. Codex support therefore needs a Responses-API-compatible endpoint (most local llama.cpp/Ollama/vLLM setups do not qualify); do not present this as working against a generic OpenAI-Chat-Completions box |
|
||||
| `gemini` | Env vars `GOOGLE_GEMINI_BASE_URL` + `GEMINI_API_KEY` + `GEMINI_MODEL`; CLI needs a restart to pick them up | **Confirmed BROKEN against llama.cpp/llama-swap, unresolved after real investigation.** Setting `GOOGLE_GEMINI_BASE_URL` makes gemini-cli internally select an `AuthType.GATEWAY` auth path (undocumented — inferred from behaviour) with validation requirements distinct from every normal auth mode; a real run against llama-swap fails with `Invalid auth method selected` regardless of what key/format is supplied. Tried and all failed: a Google-format dummy API key, `GOOGLE_GENAI_USE_VERTEXAI=false`, a `GEMINI_DEFAULT_AUTH_TYPE` override, and hand-writing `settings.json` directly. `--skip-trust` was a real, separate fix (without it a trust-folder check silently overrides `--approval-mode yolo` back to `default`) but does not touch this auth failure. Documented as an open gap, not shipped as working — the registry entry and injection code exist and are exercised by the test script, but end-to-end gemini support needs upstream investigation of `GATEWAY` AuthType before it can be called done |
|
||||
| `pi` | Config file `~/.pi/agent/models.json` with a custom provider whose `models` is an **array** of `{id}` objects (not an object keyed by id) plus `authHeader: true`. Redirected via the child process's own `HOME` env var, isolated per test/session — **not** `PI_CONFIG_DIR`, which does nothing for pi (grepped pi's entire bundled JS source: the string appears nowhere) | **Verified end-to-end** against a real llama-swap server — real "hello world" reply came back. Two real bugs found and fixed before this worked: (1) `PI_CONFIG_DIR` is not read by pi at all — pi hardcodes `~/.pi/agent/models.json` with no dedicated override, so the actual redirect has to be the child process's `HOME`; (2) `models` must be an array of `{id}` objects per pi's own bundled `docs/models.md`, not an object keyed by model id (silently loaded zero models). Also requires an explicit `--model custom/<id>` on invocation — without it pi falls back to its own default provider and fails with "No API key found for the selected model" |
|
||||
| `grok` | TOML `config.toml`: a fixed `[model.codeman-custom]` block (`base_url`, `env_key` naming an env var the key rides in, never a literal TOML field) written to an isolated dir via `GROK_HOME`. Invoked with `-m codeman-custom` | **Verified end-to-end** against a real llama-swap server — real "hello world" reply came back. The ORIGINAL recipe in this table (env vars `GROK_BASE_URL`/`XAI_API_KEY`/`GROK_MODEL`) was flat-out **wrong**, not just unverified: it produced "Not signed in" against a real binary. Grok's real mechanism, confirmed against xAI's own docs and a live binary, is a `config.toml` with a `[model.<name>]` block, redirected via `GROK_HOME`; the key still rides as an env var (`XAI_API_KEY` via `env_key`), just referenced from the TOML rather than read directly |
|
||||
| `deepseek` | Reuse the **existing** `DEEPSEEK_BASE_URL` + `DEEPSEEK_API_KEY` keys (already declared in `stock.ts`). Only `DEEPSEEK_BASE_URL` is in `privilegedEnvKeys` — `DEEPSEEK_API_KEY` deliberately stays clamp-exempt, since a non-granted owner supplying their OWN key removes privilege rather than granting it (adding it to the clamp list was a real regression, caught by `test/deepseek-mode.test.ts` and fixed before merge). No model-selection var — dsh model is a profile composition entry, not a flag/env var | **Confirmed reaching the server, but failing — unresolved.** A real run against llama-swap returns `dsh: HTTP_404: DeepSeek API error (HTTP 404)` consistently (confirmed the env vars are read: the request reaches the network rather than failing locally). Root cause not identified — plausible explanation by analogy with codex's Responses-API gap is that `dsh --profile headless` expects DeepSeek's official API response shape/path structure rather than a generic OpenAI-compatible `/v1/chat/completions` endpoint, but this was not confirmed by reading dsh's own bundled source (unlike pi/grok, where that grep resolved the question directly). Documented as best-effort/unknown, not shipped as verified working |
|
||||
| `omp` | Config file `~/.omp/agent/models.yml` with the same array-shaped `models` + `authHeader: true` fix as pi. Redirected via `HOME`, same reasoning as pi (`PI_CONFIG_DIR` does not relocate omp's config either, despite an earlier CLAUDE.md note claiming it does) | **Verified end-to-end** against a real llama-swap server — real "hello world" reply came back, after applying the same two fixes as pi (array-shaped `models`, `HOME`-redirect instead of `PI_CONFIG_DIR`) plus an explicit `--model custom/<id>` on invocation. Unverified against omp's own official docs (none are bundled in the install), but empirically confirmed working live |
|
||||
| `antigravity` | No CLI/env/config mechanism found — Antigravity's docs describe only a GUI settings panel, and explicitly say a custom endpoint "cannot currently" become the core reasoning model. **Not implemented**; toolbar entry stays disabled for this mode with an explanatory tooltip | No known mechanism |
|
||||
| CLI | Mechanism | Confidence |
|
||||
| ------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `claude` | Env vars: `ANTHROPIC_BASE_URL`, `ANTHROPIC_API_KEY`, `ANTHROPIC_DEFAULT_SONNET_MODEL`/`_HAIKU_MODEL`/`_OPUS_MODEL` (all set to the chosen model/deployment name) | **Verified end-to-end** against a real llama-swap server — a real "hello world" reply came back. ⚠️ Non-interactive (`-p`) invocations also fire an async session-title-generation call that reuses `ANTHROPIC_DEFAULT_HAIKU_MODEL` and validates it against Claude Code's OWN internal recognized-model list, printing `[claude-code:unrecognized_model]` and, in `-p` mode, hanging the whole invocation rather than just warning. `--settings '{"autoTitle":false}'` does NOT stop this (confirmed); `--bare` does (the warning still prints, but the real prompt runs) — but `--bare` ALSO disables hooks, LSP, plugin sync, and CLAUDE.md auto-discovery, so it is only safe for the standalone one-shot test script, NEVER for a real interactive Codeman session (which depends on hooks for idle detection, trust-dialog auto-accept, etc. — see the External CLI modes section of CLAUDE.md). Whether an INTERACTIVE claude session with a custom model hits the same hang (vs. just a background warning) is untested and should be checked before calling chunk 5/6 done for claude |
|
||||
| `opencode` | `OPENCODE_CONFIG_CONTENT` env var (already a registry mechanism, `stock.ts:342`) holding a JSON blob: `{"provider":{"custom":{"options":{"baseURL":...,"apiKey":...},"models":{"<name>":{}}}},"model":"custom/<name>"}` | **Verified by user** |
|
||||
| `codex` | TOML `config.toml`: top-level `model = "<id>"` + `[model_providers.custom]` (`base_url`, `env_key` naming an env var the real API key rides in — never a literal TOML field, since codex's schema has no such field). Written to an isolated dir via `CODEX_HOME` (`stock.ts:405-415`) so the user's own `~/.codex/config.toml` is never touched | **Config STRUCTURE verified** against a real codex binary (an earlier `[model].default` table shape was rejected: "invalid type: map, expected a string" — caught live). **Protocol picture more nuanced than a flat break, re-verified live twice on 2026-09-17 against a llama-swap deployment that DOES answer `/v1/responses`** (an earlier test's `Reconnecting...`/`high demand` failure does not reproduce against every llama-swap setup): a plain, no-tool-call chat turn (`codex exec 'reply with just OK'`) returned a real reply. But a real tool-call attempt (`run the shell command: echo hello`) came back as an `agent_message` TEXT item — the tool-call JSON printed as the model's answer, not a `function_call` item codex would actually execute (confirmed via `codex exec --json`'s raw event stream: `item.completed`/`agent_message`, never `function_call`). Since tool execution is what makes codex a coding agent at all, this remains **not usable for real work**, just with a different, more specific failure mode than previously documented — still do not present this as working. Separately, EVERY custom-endpoint codex session also prints `warning: Model metadata for '<id>' not found. Defaulting to fallback metadata...` on launch (confirmed harmless — the successful plain-text reply above still had it): codex's per-model metadata (reasoning tiers, system-prompt templates, context-window figures) comes from `models_cache.json`, a LOCAL CACHE of OpenAI's own hosted model catalog that a custom model can never appear in by construction. No config.toml override exists for it, and the isolated `CODEX_HOME` never gets a `models_cache.json` written into it at all (confirmed: inspected a live, actively-used isolated dir — codex evidently can't reach OpenAI's catalog endpoint for this session and just falls back silently every time, with no file left behind to fix or clean up). Fabricating a fake catalog entry to suppress the warning would mean copying the _shape_ of OpenAI's own proprietary schema — including their real per-model system-prompt content, visible in a genuine `models_cache.json` — for a warning confirmed to have no effect on the actual (broken) tool-calling outcome; not worth building |
|
||||
| `gemini` | Env vars `GOOGLE_GEMINI_BASE_URL` + `GEMINI_API_KEY` + `GEMINI_MODEL`; CLI needs a restart to pick them up | **Confirmed BROKEN against llama.cpp/llama-swap, unresolved after real investigation.** Setting `GOOGLE_GEMINI_BASE_URL` makes gemini-cli internally select an `AuthType.GATEWAY` auth path (undocumented — inferred from behaviour) with validation requirements distinct from every normal auth mode; a real run against llama-swap fails with `Invalid auth method selected` regardless of what key/format is supplied. Tried and all failed: a Google-format dummy API key, `GOOGLE_GENAI_USE_VERTEXAI=false`, a `GEMINI_DEFAULT_AUTH_TYPE` override, and hand-writing `settings.json` directly. `--skip-trust` was a real, separate fix (without it a trust-folder check silently overrides `--approval-mode yolo` back to `default`) but does not touch this auth failure. Documented as an open gap, not shipped as working — the registry entry and injection code exist and are exercised by the test script, but end-to-end gemini support needs upstream investigation of `GATEWAY` AuthType before it can be called done |
|
||||
| `pi` | Config file `~/.pi/agent/models.json` with a custom provider whose `models` is an **array** of `{id}` objects (not an object keyed by id) plus `authHeader: true`. Redirected via the child process's own `HOME` env var, isolated per test/session — **not** `PI_CONFIG_DIR`, which does nothing for pi (grepped pi's entire bundled JS source: the string appears nowhere) | **Verified end-to-end** against a real llama-swap server — real "hello world" reply came back. Two real bugs found and fixed before this worked: (1) `PI_CONFIG_DIR` is not read by pi at all — pi hardcodes `~/.pi/agent/models.json` with no dedicated override, so the actual redirect has to be the child process's `HOME`; (2) `models` must be an array of `{id}` objects per pi's own bundled `docs/models.md`, not an object keyed by model id (silently loaded zero models). Also requires an explicit `--model custom/<id>` on invocation — without it pi falls back to its own default provider and fails with "No API key found for the selected model" |
|
||||
| `grok` | TOML `config.toml`: a fixed `[model.codeman-custom]` block (`base_url`, `env_key` naming an env var the key rides in, never a literal TOML field) written to an isolated dir via `GROK_HOME`. Invoked with `-m codeman-custom` | **Verified end-to-end** against a real llama-swap server — real "hello world" reply came back. The ORIGINAL recipe in this table (env vars `GROK_BASE_URL`/`XAI_API_KEY`/`GROK_MODEL`) was flat-out **wrong**, not just unverified: it produced "Not signed in" against a real binary. Grok's real mechanism, confirmed against xAI's own docs and a live binary, is a `config.toml` with a `[model.<name>]` block, redirected via `GROK_HOME`; the key still rides as an env var (`XAI_API_KEY` via `env_key`), just referenced from the TOML rather than read directly |
|
||||
| `deepseek` | Reuse the **existing** `DEEPSEEK_BASE_URL` + `DEEPSEEK_API_KEY` keys (already declared in `stock.ts`), now with `appendV1Suffix: true` (see confidence). Only `DEEPSEEK_BASE_URL` is in `privilegedEnvKeys` — `DEEPSEEK_API_KEY` deliberately stays clamp-exempt, since a non-granted owner supplying their OWN key removes privilege rather than granting it (adding it to the clamp list was a real regression, caught by `test/deepseek-mode.test.ts` and fixed before merge). No model-selection var — dsh model is a profile composition entry, not a flag/env var | **Root cause of the original `HTTP_404` found and fixed, by reading dsh's own bundled source — the same bar pi/grok's fixes were held to.** Installed `@deepseek-ai/dsh` (all its real published dependencies) into a scratch directory purely to read `@deepseek-ai/dsh-llm-deepseek/lib/index.js`: it builds its request as `fetch(\`${connection.baseURL}/chat/completions\`, ...)`with`baseURL`read straight from`DEEPSEEK_BASE_URL`(or defaulting to DeepSeek's real public API root,`https://api.deepseek.com`, which also carries no `/v1`) — no `/v1` insertion of dsh's own, unlike the OpenAI-SDK convention this recipe originally assumed. llama-swap/llama.cpp only ever serves the OpenAI-conventional `/v1/chat/completions`. Confirmed live: `POST <baseUrl>/chat/completions` → `404`, `POST <baseUrl>/v1/chat/completions` → `200`, on the exact same endpoint — and dsh's own error-message template, `DeepSeek API error (HTTP ${status})`, reproduces the originally reported `dsh: HTTP_404: DeepSeek API error (HTTP 404)` precisely. Fixed by adding `appendV1Suffix` (env kind only, deepseek's entry alone — claude/gemini must NOT get it, since claude was already confirmed working against the unmodified `baseUrl`), which runs `endpoint.baseUrl` through the same `withV1Suffix()` helper `configDir`-kind CLIs already use. ⚠️ Not yet re-run end-to-end with a real `dsh` binary — no install available in this environment (no npm-installed CLI binary in `PATH`, and the `codeman-test-picker` container doesn't bundle it either); the fix is source-confirmed and live-verified at the HTTP level, but a genuine "hello world" reply through `dsh` itself is the remaining step before promoting this to **verified** alongside claude/opencode/pi/grok/omp |
|
||||
| `omp` | Config file `~/.omp/agent/models.yml` with the same array-shaped `models` + `authHeader: true` fix as pi. Redirected via `HOME`, same reasoning as pi (`PI_CONFIG_DIR` does not relocate omp's config either, despite an earlier CLAUDE.md note claiming it does) | **Verified end-to-end** against a real llama-swap server — real "hello world" reply came back, after applying the same two fixes as pi (array-shaped `models`, `HOME`-redirect instead of `PI_CONFIG_DIR`) plus an explicit `--model custom/<id>` on invocation. Unverified against omp's own official docs (none are bundled in the install), but empirically confirmed working live |
|
||||
| `antigravity` | No CLI/env/config mechanism found — Antigravity's docs describe only a GUI settings panel, and explicitly say a custom endpoint "cannot currently" become the core reasoning model. **Not implemented**; toolbar entry stays disabled for this mode with an explanatory tooltip | No known mechanism |
|
||||
|
||||
Everything web-researched-but-unverified gets implemented but must be
|
||||
smoke-tested against real installs of those CLIs before being called done —
|
||||
@@ -208,6 +208,14 @@ extra per-model configuration on Codeman's side at all.
|
||||
|
||||
### 4. Toolbar UI
|
||||
|
||||
> **Superseded.** This section describes the toolbar-button design as originally
|
||||
> planned. What actually shipped is a Run-menu picker instead: one generated entry
|
||||
> per (capable harness, saved endpoint) pair directly in the existing `#runModeMenu`
|
||||
> dropdown, rather than a separate `#customModelBtn`/`#customModelMenu` surface. See
|
||||
> [`docs/custom-model-endpoints.md`](custom-model-endpoints.md#the-run-menu-picker)
|
||||
> for the current design; the sections below (session-restart mechanics, security)
|
||||
> remain accurate regardless of which UI calls the underlying route.
|
||||
|
||||
- New header/toolbar button (e.g. `#customModelBtn`, `btn-toolbar
|
||||
btn-custom-model`), marker-hidden by default (`btn-custom-model--hidden`)
|
||||
and revealed by `applyHeaderVisibilitySettings()` only when
|
||||
@@ -344,8 +352,12 @@ pure unit tests and the live manual checks in Verification:
|
||||
up automatically with zero edits to the script). Already run to
|
||||
completion against the author's llama-swap server (a LAN address,
|
||||
inside a `codeman/agent:llm-test` Docker image with all 9 CLI binaries):
|
||||
claude/opencode/pi/grok/omp **PASS**, codex **FAILs as expected**
|
||||
(Responses-API protocol gap, not a bug), gemini/deepseek **UNCONFIRMED**
|
||||
claude/opencode/pi/grok/omp **PASS**, codex **partially works and still
|
||||
isn't usable** (plain chat succeeds against a llama-swap deployment that
|
||||
answers `/v1/responses`, but a real tool-call attempt comes back as
|
||||
inert text rather than an executable `function_call` — see the
|
||||
confidence table row for the full, re-verified picture), gemini/deepseek
|
||||
**UNCONFIRMED**
|
||||
(reach the server, fail for undiagnosed reasons — see their table rows),
|
||||
antigravity **SKIP** (no mechanism). Re-run this against a real cloud
|
||||
endpoint (e.g. an Azure AI Foundry deployment) once one is available, to
|
||||
|
||||
+434
-18
@@ -11,20 +11,22 @@ company gateway) — anything answering `GET /v1/models` and
|
||||
recipe confidence table, and security reasoning:
|
||||
[`custom-model-endpoints-plan.md`](custom-model-endpoints-plan.md).
|
||||
|
||||
> **Status**: backend is implemented and tested (registry capability, the
|
||||
> injection engine, the endpoint store + discovery route, the session
|
||||
> restart route). The toolbar picker / settings UI described below as the
|
||||
> intended surface is **not yet built** — until it lands, use the HTTP API
|
||||
> directly (examples below). Antigravity has no known custom-endpoint
|
||||
> mechanism and is not supported.
|
||||
> **Status**: fully wired end to end — registry capability, the injection
|
||||
> engine, the endpoint store + discovery route, both the restart-in-place
|
||||
> apply route (Claude) and the one-shot quick-start launch path (every
|
||||
> other supported harness), a settings-panel CRUD surface, and the Run-menu
|
||||
> picker described below. Antigravity has no known custom-endpoint
|
||||
> mechanism and is not supported. The HTTP API (examples below) still works
|
||||
> directly and is what the picker itself calls under the hood.
|
||||
|
||||
## Turning it on
|
||||
|
||||
App Settings → Agents & CLIs → **Custom Model Endpoints** (synced setting
|
||||
`customModelEndpointsEnabled`, default **OFF**). Until the toolbar picker
|
||||
lands, nothing reads this setting: the HTTP routes below work whether it is
|
||||
on or off, and it exists now only so the picker has a switch to hang off
|
||||
when it ships. The API equivalent:
|
||||
App Settings → Models → **Custom model endpoints** (synced setting
|
||||
`customModelEndpointsEnabled`, default **OFF**). Turning it on does two
|
||||
things: it reveals the endpoint list/add/edit/discover panel in that same
|
||||
settings section, and it makes the Run menu offer a generated entry per
|
||||
(harness, endpoint) pair — see "The Run-menu picker" below. The API
|
||||
equivalent:
|
||||
|
||||
```bash
|
||||
curl -sk -X PUT https://localhost:3000/api/settings \
|
||||
@@ -34,6 +36,9 @@ curl -sk -X PUT https://localhost:3000/api/settings \
|
||||
|
||||
## Adding an endpoint
|
||||
|
||||
Via App Settings → Models → Custom model endpoints → **+ Add endpoint**, or
|
||||
directly:
|
||||
|
||||
```bash
|
||||
curl -sk -X POST https://localhost:3000/api/model-endpoints \
|
||||
-H 'Content-Type: application/json' \
|
||||
@@ -62,7 +67,201 @@ configured, `PUT`/`DELETE /api/model-endpoints/:id` update or remove one.
|
||||
Endpoint management is admin-only in multi-user mode, same as remote/docker
|
||||
hosts — these are machine-level infra, not per-user settings.
|
||||
|
||||
## Applying a model to a session
|
||||
**Context length is discovered too, opportunistically and safely.** The plain
|
||||
`GET /v1/models` response has no context-window field. Discovery only ever
|
||||
looks for one for a model llama-swap's own response already reports
|
||||
`status.value === "loaded"` for — never for an unloaded one, because
|
||||
llama-swap treats `?model=` as a routing hint and asking about a model that
|
||||
isn't loaded risks triggering an actual (slow, GPU-swapping) load as a side
|
||||
effect of what should be read-only discovery. A server with no `status` field
|
||||
on any entry at all (not llama-swap) gets no context-length enrichment,
|
||||
rather than guessing. A model's previously-learned context length survives a
|
||||
later cycle where it wasn't the loaded one; it's dropped only once the model
|
||||
disappears from the endpoint's list entirely. Stored per model in
|
||||
`modelContextLengths` and applied automatically (see "Applying a model to a
|
||||
session" below) so a CLI that would otherwise assume a large default context
|
||||
window for an unrecognized model id stops silently overflowing a much
|
||||
smaller real one.
|
||||
|
||||
**Where that number actually comes from matters, and got this wrong once
|
||||
already.** The first cut read it from llama.cpp's own
|
||||
`GET /props?model=<id>` (`n_ctx`) — plausible, and it worked in testing, but
|
||||
confirmed live to be actively WRONG for a `--fit-ctx`-launched llama-swap
|
||||
backend: `/props` reported `n_ctx: 154112` for a model llama-swap itself had
|
||||
launched with `--fit-ctx 16384`, and the real server then refused a request
|
||||
right at that real 16384-token limit — `/props`'s `n_ctx` appears to report
|
||||
the model's theoretical/trained maximum there, not the runtime-configured
|
||||
one. Discovery now parses the REAL configured size straight out of
|
||||
llama-swap's own launch command instead (`GET /running`'s `cmd` field —
|
||||
`--fit-ctx <N>` first, then the plain llama.cpp `-c`/`--ctx-size` a
|
||||
hand-written command might use), and only falls back to the `/props` probe
|
||||
when `cmd` states no recognizable flag at all.
|
||||
|
||||
**File size is discovered too, when the server states one.** llama-swap
|
||||
writes a GB figure into an auto-discovered model's own `description`
|
||||
(`"Auto-discovered 16.35 GB - parameters auto-fitted by llama.cpp"`), parsed
|
||||
into `modelSizesGB` — unlike context length, this needs no `/props` probe
|
||||
(the figure is right there in the `/v1/models` response) and so is populated
|
||||
for every model regardless of loaded state. A hand-configured profile's own
|
||||
description has no such figure and correctly gets no entry, never a guess.
|
||||
Used only to label the Run-menu picker's "loading model" banner (e.g.
|
||||
"Loading qwen3.8-27b-ud-q4_k_xl (16.4 GB) on llama-swap..."); never anything
|
||||
a server-side check relies on.
|
||||
|
||||
**The loading banner is unbounded by design, and says so — no countdown, no
|
||||
automatic give-up.** An earlier version scaled an expected-time estimate and
|
||||
a timeout off the model's file size and auto-closed the session once that
|
||||
elapsed, but a real load's actual duration depends on hardware this feature
|
||||
has no way to know (VRAM, storage speed, whatever else is contending for the
|
||||
GPU) — any fixed number was a guess dressed up as a fact, and a model that
|
||||
genuinely takes 10+ minutes on slower hardware would just get killed
|
||||
mid-load by its own display. The banner now says outright that it can take a
|
||||
while depending on hardware and model size, polls
|
||||
`GET /api/model-endpoints/:id/running-status` every second for as long as it
|
||||
takes, and carries a **Cancel** button (rendered on the banner itself) that
|
||||
ends the wait and closes the session the load was for — the user's own call
|
||||
on when it's taking too long, not a fixed number baked into the client.
|
||||
|
||||
**The banner's second line is the real backend log line, not a guess.**
|
||||
llama-swap's `GET /api/events` SSE stream carries the actual `llama-server`
|
||||
process's own stdout — `load_model: loading model '<path>'`,
|
||||
`llama_server: model loaded`, tokenizer warnings, all of it — tagged
|
||||
`source: "upstream"`, distinct from llama-swap's own `source: "proxy"`
|
||||
request-access lines. `running-status`'s response now includes `logLine`
|
||||
(via `getLatestLlamaSwapLogLine`), and the banner shows it on its own line
|
||||
under the disclaimer, e.g. "llama.cpp: load_model: loading model '...'" —
|
||||
confirmed live end-to-end through a real forced swap, sequentially showing
|
||||
the model path, a tokenizer warning, then staying on whatever llama.cpp last
|
||||
printed once the load goes quiet (never cleared back to blank). ⚠️
|
||||
**`GET /logs` — the endpoint this feature's own first cut was built
|
||||
against — turns out to carry ONLY llama-swap's own proxy request-access
|
||||
log.** Confirmed live it never showed a single backend line, even seconds
|
||||
after a real, verified model swap; `/api/events`'s `logData` frames are the
|
||||
only source that actually has it, and its own `source` field (`upstream` vs
|
||||
`proxy`) is what `getLatestLlamaSwapLogLine` filters on. One `/api/events`
|
||||
connection is held open per endpoint and reused across every session
|
||||
watching a load on it (confirmed live to stay open indefinitely, unlike
|
||||
`/logs`, which closes after a fixed ~100KB), idle-closed after 30s of nobody
|
||||
polling it (`pruneIdleLlamaSwapLogTails`, same 20s sweep as the
|
||||
swap-displacement check below).
|
||||
|
||||
`defaultModelId` names which discovered model the picker pre-marks for that
|
||||
endpoint — the settings panel's Edit form exposes it as a select populated
|
||||
from the endpoint's own discovered `models`, and the route refuses a value
|
||||
that isn't one of them. It is applied automatically only when the endpoint
|
||||
has exactly one discovered model (nothing to choose); with two or more it
|
||||
is a pre-selection in the model-picker dialog below, never a silent default.
|
||||
Re-discovering drops a default that no longer appears in the fresh list
|
||||
rather than carrying an invalid one forward.
|
||||
|
||||
**Model lists refresh themselves.** A background sweep (`server.ts`,
|
||||
`CUSTOM_MODEL_REDISCOVER_INTERVAL_MS`, every 5 minutes) re-discovers every
|
||||
saved endpoint the same way the manual `POST .../discover-models` route
|
||||
does, best-effort per endpoint — one being unreachable on a given cycle
|
||||
never blocks the others. Off under `npm test`, same reasoning as the Codex
|
||||
plan-usage poll it sits beside: no real network to hit, no server instance
|
||||
to keep the timer alive for.
|
||||
|
||||
## The Run-menu picker
|
||||
|
||||
With the setting on and at least one endpoint carrying a discovered model,
|
||||
the toolbar's Run dropdown grows a **Custom Endpoints** section: one entry
|
||||
per (harness that can redirect to a custom endpoint, saved endpoint) pair,
|
||||
e.g. "Claude Code (llama.cpp)". The harness list is read off the CLI
|
||||
registry's own `capabilities.customModelInjection` at page render
|
||||
(`window.__codemanCustomModelClis`, `server.ts`) — never a hardcoded id list
|
||||
in the frontend — so a CLI whose injection recipe lands later shows up with
|
||||
no frontend change, and Antigravity (`unsupported`) never does.
|
||||
|
||||
Picking an entry re-fetches the endpoint (`selectCustomModelEntry()`,
|
||||
`session-ui.js`) rather than trusting anything cached from the dropdown's
|
||||
own render — the model list can have changed via the 5-minute sweep above
|
||||
or a settings-panel edit since the menu opened. With exactly one discovered
|
||||
model it runs straight away; with two or more, a small modal
|
||||
(`#customModelPickModal`) lists them and asks which one to use for this
|
||||
launch, with the endpoint's `defaultModelId` marked but not auto-chosen —
|
||||
the point of asking is letting one launch deliberately differ from the
|
||||
saved default, not just confirming it.
|
||||
|
||||
The modal promotes exactly one row to the top of the list rather than
|
||||
always showing raw discovery order, so the zero-wait choice is the one
|
||||
under your thumb:
|
||||
|
||||
- **"Currently loaded"** — a model from this host's own list that
|
||||
llama-swap reports `ready` right now, queried via
|
||||
`GET /api/model-endpoints/:id/running-status`. Bounded client-side to
|
||||
~800ms (`Promise.race`), on top of the route's own 5s server-side
|
||||
timeout, so an endpoint that is asleep or firewalled cannot leave the
|
||||
modal invisible for the full 5s after the Run menu has already closed.
|
||||
- **"Last used"** — shown only when nothing is currently loaded: the model
|
||||
actually launched last for this exact (harness, endpoint) pair, read
|
||||
from the per-device `codeman:customModelLastUsed:<mode>:<endpointId>`
|
||||
localStorage key. Written by `_runCustomModelEntryViaRestart` (claude)
|
||||
and `_quickStartWithCustomModelConfirm` (every one-shot launch; the
|
||||
`runCustomModelEntry` entry point itself only dispatches between the
|
||||
two) only once the model is actually applied, never on the mere click —
|
||||
declining the context-window warning means this exact model cannot work
|
||||
with this CLI at all, so promoting it next time would be actively wrong,
|
||||
not just premature.
|
||||
|
||||
Neither tag reorders anything past that one promoted row. The "Default"
|
||||
pill is a separate span, not a third value of the same slot: a promoted
|
||||
row that is also the endpoint's `defaultModelId` shows both tags (on a
|
||||
single-purpose GPU box that is the common case, and an exclusive slot
|
||||
silently dropped the Default marking for exactly that row), and a row
|
||||
with neither promotion nor default shows no tag at all.
|
||||
|
||||
**How the launch itself applies the endpoint depends on the harness.** For
|
||||
opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP (`runCustomModelEntry` →
|
||||
`_runCustomModelEntryOneShot`), the endpoint/model is folded into the SAME
|
||||
`POST /api/quick-start` call that creates the session (`customModel` field),
|
||||
so the session launches directly on the endpoint — no restart, no visible
|
||||
relaunch. Claude (`_runCustomModelEntryViaRestart`) still uses the original
|
||||
two-step design: the launch runs a single native session exactly the way its
|
||||
own Run-menu entry would, then **waits for the new session to go idle**
|
||||
(`GET .../wait?until=idle`, bounded at 20s — a normal 200 either way, never
|
||||
an error, per the wait endpoint's own contract) before applying the endpoint
|
||||
via the restart route below. That wait exists because a freshly launched CLI
|
||||
reports itself as `busy` for its own startup (a boot spinner, a
|
||||
workspace-trust check) well before the apply call would otherwise reach it,
|
||||
and the apply route correctly refuses to restart a session mid-turn — a
|
||||
fresh boot looks exactly like one from the outside. A session still busy
|
||||
after the wait reaches the apply call anyway and gets that route's own
|
||||
honest `SESSION_BUSY` error, now visible as a sticky toast with a close
|
||||
button rather than a generic message that vanished in three seconds. Claude
|
||||
stays on this path because its own restart (`--resume`-based, keeping the
|
||||
conversation) is far less jarring than the other seven's, and `runClaude()`'s
|
||||
multi-tab launch and docker-config-drift confirm/retry loop make folding it
|
||||
into the one-shot path separate work. It is a
|
||||
one-off "try this endpoint" action, not a sticky mode: the plain Run button
|
||||
still means "this harness, native cloud" afterward. Entries are hidden
|
||||
entirely for a remote or Docker active case, since the apply route refuses
|
||||
both (see the next section).
|
||||
|
||||
## Launching directly on an endpoint (no restart)
|
||||
|
||||
```bash
|
||||
curl -sk -X POST https://localhost:3000/api/quick-start \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"caseName": "myapp", "mode": "codex", "customModel": {"endpointId": "llama-box", "modelId": "qwen3"}}'
|
||||
```
|
||||
|
||||
`POST /api/quick-start`'s `customModel` field (`{endpointId, modelId,
|
||||
confirmed?}`) computes the same injection the restart route below does, but
|
||||
BEFORE the session exists — the session is minted its own id up front
|
||||
(`crypto.randomUUID()`), the injection (env vars, and for a `configDir`-kind
|
||||
CLI, the written config file) targets that real id, and the session launches
|
||||
already pointed at the endpoint. No restart, because there was never a
|
||||
native-backend launch to restart away from. Runs the same llama-swap
|
||||
conflict check as the restart route (below) — a `409`-shaped
|
||||
`{requiresConfirmation, currentlyLoadedModel, affectedSessions}` response
|
||||
with no session created, resolved by retrying with `confirmedSwap: true` — and
|
||||
is refused the same way for a remote or Docker case. This is what the
|
||||
Run-menu picker uses for opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP;
|
||||
Claude still uses the restart route below (see "The Run-menu picker" above
|
||||
for why).
|
||||
|
||||
## Applying a model to an ALREADY-RUNNING session
|
||||
|
||||
```bash
|
||||
curl -sk -X POST https://localhost:3000/api/sessions/<sessionId>/custom-model \
|
||||
@@ -84,6 +283,179 @@ since for those three the config file alone does not switch the model.
|
||||
reattaches the durable remote/in-container tmux rather than relaunching the
|
||||
agent, so the selection would report success and change nothing.
|
||||
|
||||
**Claude gets two more env vars when known/applicable, both declared on its
|
||||
registry entry (`contextLengthVar`/`configDirVar`), not hardcoded here:**
|
||||
|
||||
- `CLAUDE_CODE_MAX_CONTEXT_TOKENS` is set to `modelId`'s discovered context
|
||||
length (see the discovery section above) whenever one is known. Without
|
||||
it, Claude Code assumes a large (200k) window for any unrecognized custom
|
||||
model id and never compacts, which reliably overflows a much smaller real
|
||||
local context — confirmed live: a stock ~33.7K-token system prompt against
|
||||
a 16384-token llama-swap model failed with `exceeds the available context
|
||||
size`. No entry for the model in `modelContextLengths` means the var is
|
||||
simply omitted, never a guess. ⚠️ **This var only affects when Claude
|
||||
Code compacts conversation _history_ — it cannot fix a model whose real
|
||||
context is smaller than Claude Code's own fixed per-turn overhead**
|
||||
(system prompt + tool schemas, empirically ~36.4K tokens, confirmed live
|
||||
via an `in:0 out:0` failure on the very first message, before any
|
||||
history exists to compact). No context-length declaration changes that
|
||||
fixed overhead, so a model below the safe floor fails outright on
|
||||
message one regardless of what this var says. See "Context-window floor
|
||||
warning" below for how Codeman catches this case before launching
|
||||
instead of after.
|
||||
- `CLAUDE_CONFIG_DIR` is pointed at the same isolated per-session directory
|
||||
the `configDir`-kind CLIs use (empty, no files written into it), so the
|
||||
injected `ANTHROPIC_API_KEY` never shares a directory with a stored
|
||||
claude.ai OAuth login. Claude Code still prints "Both claude.ai and
|
||||
ANTHROPIC_API_KEY set" when the two coexist in the same config directory —
|
||||
cosmetic (confirmed live: the API key wins for actual requests either way,
|
||||
visible in the terminal's own `API Usage Billing` line) but worth
|
||||
eliminating rather than living with. The directory's `projects`
|
||||
subdirectory is symlinked (a junction on Windows) back to the real
|
||||
`~/.claude/projects` so the response viewer, subagent windows and Read My
|
||||
Mind keep working for that session — the same trade-off and fix documented
|
||||
for a manually-set `CLAUDE_CONFIG_DIR` in
|
||||
[`docs/wiki/Agent-CLIs.md`](wiki/Agent-CLIs.md), just applied
|
||||
automatically here. Best-effort: a platform that refuses the symlink keeps
|
||||
the pre-existing blind-response-viewer side effect rather than failing the
|
||||
whole custom-model apply over it. ⚠️ **This relocates the whole `.claude`
|
||||
tree, not just transcripts**: a custom-model Claude session also loses the
|
||||
user's global `settings.json`, user-level skills (the codeman agent skill
|
||||
included), user-level agents and commands, and the MCP servers configured
|
||||
in `~/.claude.json` — none of those are symlinked back, only `projects` is.
|
||||
A fine trade for "point this session at my local llama.cpp," but worth
|
||||
knowing before it surprises you mid-session.
|
||||
|
||||
**That isolated directory needed one more fix to actually be usable
|
||||
non-interactively.** An otherwise-empty `CLAUDE_CONFIG_DIR` has none of a
|
||||
real profile's prior "Detected a custom API key — use it?" approvals, so
|
||||
without more, Claude Code stops and asks that on _every single launch_ —
|
||||
confirmed live, and with nobody at a TTY to answer, its own default answer
|
||||
("No") silently refuses the very key this feature just injected, which
|
||||
looks like the endpoint being ignored entirely. `customModelInjection`'s
|
||||
`apiKeyTrustFile` (`{ relPath: '.claude.json', shape:
|
||||
'claude-api-key-responses' }` on claude's entry) pre-seeds that exact
|
||||
approval: the apply step merges `customApiKeyResponses.approved: [apiKey]`
|
||||
into `<configDir>/.claude.json`, the same field a real answered prompt
|
||||
itself writes to (confirmed against a real file after answering by hand
|
||||
once) — this answers the prompt in advance rather than bypassing it. The
|
||||
merge preserves whatever else the CLI already wrote into that file on an
|
||||
earlier launch in the same isolated directory (`userID`, `numStartups`,
|
||||
earlier approved keys), and a missing or corrupt file is treated as empty
|
||||
rather than failing the apply.
|
||||
|
||||
**A fresh `CLAUDE_CONFIG_DIR` isn't just missing that one approval — Claude
|
||||
Code treats it as a brand-new profile and replays its ENTIRE first-run
|
||||
sequence on every launch: the theme picker, the security-notes screen, the
|
||||
per-project "trust this folder?" dialog, and (running with
|
||||
`--dangerously-skip-permissions`) a one-time warning about bypassing
|
||||
permissions.** Confirmed live: none of these show up again for a real,
|
||||
already-onboarded profile, but every custom-model session gets a fresh,
|
||||
otherwise-empty isolated directory, so it saw all four every single time.
|
||||
`customModelInjection`'s `skipFirstRunPrompts` (`true` on claude's entry,
|
||||
requires `apiKeyTrustFile` since it reuses the same file) pre-seeds the
|
||||
state a real profile accumulates from answering all of that once:
|
||||
`hasCompletedOnboarding: true` and the launching session's own
|
||||
`projects[workingDir].hasTrustDialogAccepted: true` go into the same
|
||||
`<configDir>/.claude.json` the API-key approval above already merges into
|
||||
(other projects, and other fields on this session's own project entry, are
|
||||
left untouched), and `skipDangerousModePermissionPrompt: true` goes into
|
||||
`<configDir>/settings.json` — a different file, merged the same
|
||||
corrupt-tolerant way. `workingDir` is used exactly as the session was
|
||||
launched with as its cwd, never realpath'd or slash-normalized, since
|
||||
that's the literal string Claude Code itself uses as the project key.
|
||||
|
||||
**llama-swap gets two more fixes on top of the context-length/config-dir
|
||||
ones above, both from watching a real switch live.** llama.cpp only ever
|
||||
runs one model at a time; llama-swap swaps the backing process on demand,
|
||||
which can take anywhere from a few seconds to well over a minute:
|
||||
|
||||
- **The conflict check.** Both apply routes (the restart one here and the
|
||||
one-shot `POST /api/quick-start` above) call llama-swap's own
|
||||
`GET /running` first — feature-detected, so a plain llama.cpp/OpenAI-
|
||||
compatible server (no such endpoint) is simply never checked. If a
|
||||
_different_ model is currently loaded and ready, and another **live
|
||||
session's own selection** is using it, the apply returns
|
||||
`{requiresConfirmation: true, currentlyLoadedModel, affectedSessions}`
|
||||
instead of silently switching — nothing is applied or created yet.
|
||||
Retrying with `confirmedSwap: true` skips the check (the legacy `confirmed: true`
|
||||
still means both questions). Switching with nothing
|
||||
else affected proceeds immediately; this is a warning about disrupting
|
||||
another session, never a gate on the switch itself.
|
||||
- **Actually starting the load.** llama-swap has no "switch model" admin
|
||||
call — the only thing that starts a swap is a real inference request
|
||||
naming the model, and confirmed live: applying a selection alone never
|
||||
reached llama-swap at all (nothing in its own server logs), since nothing
|
||||
had actually asked it to load anything yet. Both apply routes now also
|
||||
send the smallest real request that will —
|
||||
`POST <baseUrl>/v1/chat/completions` with `max_tokens: 1` and one
|
||||
throwaway message — whenever the
|
||||
target model isn't already the one loaded and ready, fire-and-forget (its
|
||||
response is never read; `GET /api/model-endpoints/:id/running-status`,
|
||||
polled client-side, is what actually confirms readiness). The response
|
||||
also carries `modelSwapInProgress: true` in that case, which is what
|
||||
drives the Run-menu picker's own "loading model" status banner.
|
||||
|
||||
## Catching a swap after the fact
|
||||
|
||||
The conflict check above only runs at the moment a session is created or a
|
||||
model is applied — it has no way to catch a swap that happens **later**.
|
||||
Confirmed live: a session created while nothing else conflicted at that
|
||||
exact instant can still get silently displaced afterward, once a
|
||||
_different_ session's own normal use (or its own create-time load trigger)
|
||||
asks llama-swap to load something else. llama-swap has no push
|
||||
notification of its own for this, so a background sweep
|
||||
(`detectCustomModelSwapDisplacements`, `CUSTOM_MODEL_SWAP_CHECK_INTERVAL_MS`
|
||||
= 20s in `server.ts`) polls `GET /running` once per distinct endpoint that
|
||||
has at least one live custom-model session, and compares each such
|
||||
session's own `modelId` against what is actually loaded. A session whose
|
||||
model is no longer in that list gets a `custom-model:swapped-out` SSE event
|
||||
(`{sessionId, sessionName, endpointId, previousModel, currentlyLoadedModel}`),
|
||||
shown as a global toast — global rather than tied to that session's tab,
|
||||
since the whole point is telling the user before they type into it
|
||||
expecting the model they picked. Notifies **once per displacement**: the
|
||||
same de-dupe `Set` clears a session's flag once its own model is loaded and
|
||||
ready again, so a later, genuinely new displacement notifies again rather
|
||||
than the session staying silently un-notified forever after the first one.
|
||||
|
||||
## Context-window floor warning
|
||||
|
||||
Claude Code's own fixed per-turn overhead (system prompt + tool schemas,
|
||||
empirically ~36.4K tokens) can exceed a small local model's _entire_ real
|
||||
context on its own, before any conversation history exists to fill it —
|
||||
confirmed live twice, both as an `in:0 out:0` failure on the very first
|
||||
message sent. `CLAUDE_CODE_MAX_CONTEXT_TOKENS` (above) cannot fix this: it
|
||||
only governs when Claude Code compacts conversation history, and there is
|
||||
no history yet on message one. Applying such a model would look like the
|
||||
endpoint being ignored, or the wrong model being used, when in fact the
|
||||
endpoint applied correctly and the model is simply too small for this CLI.
|
||||
|
||||
Both apply routes (the restart route and the one-shot `POST
|
||||
/api/quick-start`) now check for this **before** launching or restarting
|
||||
anything, gated on the CLI's registry entry declaring a `contextLengthVar`
|
||||
(currently only claude — the check is a no-op for every other CLI by
|
||||
construction, never a hardcoded mode check). If the model's discovered
|
||||
context (`modelContextLengths`, from discovery above) is below
|
||||
`CLAUDE_MIN_SAFE_CONTEXT_TOKENS` (40000, comfortably above the measured
|
||||
~36.4K overhead), the response is `{requiresContextWarning: true, modelId,
|
||||
contextLength, minSafeContextTokens}` instead of applying — nothing is
|
||||
restarted or created yet. A context length that was never discovered at
|
||||
all skips the check entirely (nothing to compare, so it fails open rather
|
||||
than warning on every model an endpoint hasn't reported a size for).
|
||||
Retrying with `confirmedContext: true` launches anyway (the legacy `confirmed: true` still means both questions).
|
||||
|
||||
The Run-menu picker shows this as an in-app modal
|
||||
(`#customModelContextWarningModal`, matching the llama-swap conflict
|
||||
modal's look) naming the model, its discovered context, and the safe
|
||||
floor, and explaining the fix: reconfigure llama-swap to give that model
|
||||
(or a smaller one) an explicit larger context instead of relying on
|
||||
auto-fit (`--fit-ctx`), which optimizes for the biggest _model_ that fits
|
||||
rather than the biggest _context_ — e.g. adding `-c 65536` (or as large a
|
||||
`--ctx-size` as the hardware holds) to that model's llama-swap config
|
||||
entry. A smaller model at a much larger explicit context often fits in
|
||||
the same VRAM a bigger model's auto-fit context gets shrunk to make room
|
||||
for.
|
||||
|
||||
Clear back to the harness's native cloud default with:
|
||||
|
||||
```bash
|
||||
@@ -101,6 +473,14 @@ id, model and injected key NAMES are persisted, the values are re-derived
|
||||
from the endpoint store on recovery, and the pane keeps running against the
|
||||
endpoint in between because tmux retains its environment.
|
||||
|
||||
⚠️ Clearing removes injected keys **by name**, and `CLAUDE_CONFIG_DIR` is one
|
||||
of the names claude's selection injects — so a session that ALSO had
|
||||
`CLAUDE_CONFIG_DIR` set through the generic `envOverrides` field (the
|
||||
per-client-account case) loses that override on clear too, and silently
|
||||
falls back to the server's default Claude account. If you route a session
|
||||
to a specific account this way, re-apply the override after clearing a
|
||||
custom-model selection from it.
|
||||
|
||||
**New sessions always default back to the harness's native backend.** A
|
||||
custom-endpoint selection is a per-session choice, never a sticky global
|
||||
default — starting a fresh session doesn't inherit whatever the last one was
|
||||
@@ -115,17 +495,53 @@ automatically). Results:
|
||||
|
||||
- **Claude, opencode, Pi, Grok, OMP** — verified: a real "hello world" reply
|
||||
came back through the endpoint.
|
||||
- **Codex** — the config is structurally correct, but Codex only speaks the
|
||||
Responses API since Feb 2026, which llama.cpp/llama-swap don't implement.
|
||||
This is a real protocol incompatibility, not a bug here; Codex support
|
||||
needs a Responses-API-compatible endpoint.
|
||||
- **Codex** — the config is structurally correct, and against a llama-swap
|
||||
server that DOES answer `/v1/responses` (confirmed live: a plain,
|
||||
no-tool-call chat turn returned a real reply), the picture is more
|
||||
nuanced than a flat failure. A real tool-call attempt (`run the shell
|
||||
command: echo hello`) came back as `agent_message` TEXT — literally the
|
||||
tool-call JSON printed as the model's answer — instead of a
|
||||
`function_call` item Codex would actually execute (confirmed via `codex
|
||||
exec --json`'s raw event stream). So plain chat can work while the thing
|
||||
that makes Codex a coding agent — actually running commands and editing
|
||||
files — does not; treat Codex as still unreliable for real work against a
|
||||
llama.cpp/llama-swap endpoint, tool-calling gap included, not just the
|
||||
earlier-documented `wire_api` mismatch (which not every deployment hits
|
||||
the same way — some legitimately have no `/v1/responses` route at all).
|
||||
Separately, EVERY custom-endpoint Codex session prints `Model metadata
|
||||
for '<id>' not found. Defaulting to fallback metadata...` on launch —
|
||||
confirmed harmless (the reply above still came back correctly): Codex's
|
||||
model metadata (reasoning-tier options, per-model system-prompt
|
||||
templates, context-window figures) comes from `models_cache.json`, a
|
||||
local cache of OpenAI's own hosted model catalog that a custom local
|
||||
model can never appear in by construction, since it isn't one of
|
||||
OpenAI's models. There's no config.toml override for a model's metadata,
|
||||
and fabricating a fake catalog entry would mean copying the _shape_ of
|
||||
OpenAI's own proprietary schema (their per-model system-prompt content
|
||||
included) for a warning that doesn't otherwise affect behavior — not
|
||||
something to build into discovery.
|
||||
- **Gemini** — fails with `Invalid auth method selected`, traced to an
|
||||
undocumented `GATEWAY` auth path gemini-cli selects once
|
||||
`GOOGLE_GEMINI_BASE_URL` is set. Unresolved after real investigation
|
||||
(several auth workarounds were tried and ruled out); do not rely on
|
||||
Gemini support yet.
|
||||
- **DeepSeek** — the request reaches the server (env vars are read) but
|
||||
gets a consistent `HTTP_404`. Root cause not identified; best-effort only.
|
||||
- **DeepSeek** — root cause of the `HTTP_404` found and fixed. DeepSeek
|
||||
Harness's own bundled provider module (`@deepseek-ai/dsh-llm-deepseek`)
|
||||
builds its request URL as `${DEEPSEEK_BASE_URL}/chat/completions` with no
|
||||
`/v1` insertion of its own (its real public API, `https://api.deepseek.com`,
|
||||
expects the caller's base URL to already carry any needed prefix) —
|
||||
confirmed by reading its own source and, live, that
|
||||
`POST <baseUrl>/chat/completions` 404s against llama-swap while
|
||||
`POST <baseUrl>/v1/chat/completions` succeeds; the harness's own error
|
||||
template (`DeepSeek API error (HTTP ${status})`) matches the originally
|
||||
reported symptom exactly. `customModelInjection`'s new `appendV1Suffix`
|
||||
(deepseek's entry only — claude/gemini must NOT get it, since claude was
|
||||
already confirmed working against the raw `baseUrl`) fixes it by writing
|
||||
`DEEPSEEK_BASE_URL` with `/v1` appended. Not yet re-run end-to-end with a
|
||||
real `dsh` binary (no install available in this environment) — the fix
|
||||
is source-confirmed and live-verified at the HTTP level, but a real
|
||||
"hello world" reply through `dsh` itself is still outstanding before
|
||||
calling this fully verified like the harnesses above.
|
||||
- **Antigravity** — no known custom-endpoint mechanism at all; unsupported.
|
||||
|
||||
See the confidence table in `custom-model-endpoints-plan.md` for the full detail behind
|
||||
|
||||
@@ -53,7 +53,9 @@ that spawns a literal `pnpm` with no npm fallback, so without one it exits 127 w
|
||||
surfaces that same line as the install error. `npm install -g pnpm` (or
|
||||
`corepack enable pnpm`) is the fix. This is what broke the Docker agent image in
|
||||
[#352](https://github.com/Ark0N/Codeman/issues/352); the image now installs pnpm
|
||||
alongside `dsh`.
|
||||
alongside `dsh`. The Compose server image (`docker/server.Dockerfile`) does not
|
||||
ship `dsh`, since it is installed at runtime, but it does ship pnpm so the UI
|
||||
button works there too.
|
||||
|
||||
Codeman's default is `@deepseek-harness-tui/dsh-tui` because it is by a wide
|
||||
margin the most used community TUI, it is MIT, and it implements the status
|
||||
|
||||
@@ -81,6 +81,10 @@ Antigravity (`agy`) and Grok (`grok`) are the two CLIs not installed from npm (G
|
||||
|
||||
Pi's credentials are seeded per-FILE rather than as a whole directory (`auth.json`, `settings.json`, `trust.json`, `models.json`, `models-store.json` out of `~/.pi/agent`), because that directory also holds `sessions/`, `extensions/`, `skills/` and the installed package trees — gigabytes on an active host. Consequence: in-container pi sessions are invisible host-side, so `pi -c` inside a Docker case only sees that container's own history. See [`pi-integration.md`](./pi-integration.md). Grok is seeded per-file for the same reason (`auth.json`, `config.toml`, `pager.toml` out of `~/.grok`, which also holds `sessions/`, `memory/` and the ~160MB binary under `downloads/`), with the same consequence for `grok -c`. See [`grok-integration.md`](./grok-integration.md). OMP is the one CLI in this family where `sessions/` is the EXCEPTION rather than the rule: `~/.omp/agent/{config.yml,mcp.json,models.yml,settings.yml}` are seeded per-file (the dir also holds SQLite caches and `terminal-sessions/`), but `~/.omp/agent/sessions/` is shared RW like codex's, not seeded, because Codeman reads it host-side for history recovery and `--resume` pinning. See [`omp-integration.md`](./omp-integration.md).
|
||||
|
||||
The image can also carry the GitHub CLI (`gh`) and the Azure CLI (`az` + the `azure-devops` extension, in `AZURE_EXTENSION_DIR=/opt/az-extensions` so it stays out of the seeded HOME), wired into the system git config as credential helpers for github.com and dev.azure.com / *.visualstudio.com, exactly as in `docker/server.Dockerfile`. Their sign-ins are seeded per-FILE like pi's: `~/.config/gh/{hosts.yml,config.yml}` and `~/.azure/{azureProfile.json,msal_token_cache.json,service_principal_entries.json,clouds.config,config}`, never `~/.azure`'s logs, command index or extensions. A token kept in a desktop keyring, or in the encrypted MSAL cache az uses on Windows/macOS, is not in those files and does not carry. None of the three is version-pinned; the `--no-cache` rebuild recommended above is also what refreshes them. Both CLIs are opt-in and OFF by default: `CODEMAN_AGENT_IMAGE_INSTALL_GH=1` / `CODEMAN_AGENT_IMAGE_INSTALL_AZ=1` in the environment of `scripts/build-agent-image.mjs`, or of the Codeman server for its own auto-build (in the Compose deployment, `environment:` in `docker-compose.override.yml`), become the `CODEMAN_INSTALL_GH` / `CODEMAN_INSTALL_AZ` build args and put that CLI, its extension and its helper entry into the image. Unset passes nothing, so a default build's argv is unchanged and the image has neither. The sign-in seeds follow the same switches, read when a case container is created: `.config/gh` only with `CODEMAN_AGENT_IMAGE_INSTALL_GH=1`, `.azure` only with `CODEMAN_AGENT_IMAGE_INSTALL_AZ=1` (`enabledByEnv` in `CRED_STORES`), never merely because the files exist. Seeds are create-time mounts and deliberately not part of the config hash (hashing them would trip the drift gate for every case), so an existing case container picks them up only when it is recreated.
|
||||
|
||||
Set `CODEMAN_AGENT_IMAGE_GIT_USER_NAME` and `CODEMAN_AGENT_IMAGE_GIT_USER_EMAIL` together to configure the agent image's Git identity. Rebuild an existing `codeman/agent:base` with `node scripts/build-agent-image.mjs --no-cache`, then recreate Docker-case containers so they use the rebuilt image.
|
||||
|
||||
## Quickest path: one-click "Run in Docker"
|
||||
|
||||
On the **New case → Create New** tab there's a **🐳 Run in an isolated Docker container** checkbox. Checking it alone is enough: Codeman creates the case folder in `~/codeman-cases/<name>`, spins up a hardened container with sensible defaults (auto-provisioning a shared `default` host), and starts the session inside it. No host/image/network fields to fill in.
|
||||
|
||||
@@ -6,6 +6,10 @@ For the Compose configuration, environment settings, storage migration, and macv
|
||||
|
||||
The image includes Claude Code, Codex, Gemini CLI, and OpenCode. Authenticate a CLI from its Codeman session; credentials are never baked into the image.
|
||||
|
||||
CLIs installed from **App Settings → Agents & CLIs → CLI management** (DeepSeek Harness, Pi, and the other npm-based ones) go to `~/.local` on the `CODEMAN_APPDATA_PATH` mount, so they survive an image rebuild and a container recreate. Releases up to 1.33.1 installed them into the image instead, so a CLI installed from Settings on one of those has to be installed again once after the rebuild. The same applies to a hand-run `npm install -g` inside a session: it writes to the image prefix (`/opt/codeman-cli`) and is lost on the next rebuild, so use `npm install -g --prefix ~/.local <package>` instead.
|
||||
|
||||
It can also include the GitHub CLI (`gh`) and the Azure CLI (`az`) with the `azure-devops` extension, wired in as Git credential helpers, so Clone Repo and `git clone` reach private GitHub and Azure DevOps repositories once they are signed in. Both are off by default; [Turning them on](../docker/README.md#turning-them-on) shows the `docker-compose.override.yml` settings.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Docker Engine or Docker Desktop with Docker Compose v2
|
||||
@@ -67,7 +71,7 @@ If that directory was created by an earlier root-running image, change its owner
|
||||
|
||||
Codeman updates itself from **App Settings → Updates**, as it does on a bare host. The checkout mounted at `/opt/codeman` is the same directory Compose builds from, so the update's `git checkout` and rebuild land on the host and survive container recreation; the restart is the server exiting, which `restart: unless-stopped` turns into a relaunch on the new build.
|
||||
|
||||
That applies application code only. A release that changes `docker/server.Dockerfile`, `docker/docker-compose.yaml`, or adds a key to `docker/.env.example` needs the image rebuilt or the container recreated, which a container cannot do to itself. The updater detects each case and refuses with a message naming what changed; run `docker/Start-Codeman.sh` on the host to apply those.
|
||||
That applies application code only. A release that changes `docker/server.Dockerfile`, `docker/docker-compose.yaml`, or adds a key to `docker/.env.example` needs the image rebuilt or the container recreated, which a container cannot do to itself. The updater detects each case and refuses with a message naming what changed; run `docker/Start-Codeman.sh` on the host to apply those. For a major update, or a base-image change `Start-Codeman.sh` does not fully pick up, `docker/Update-Codeman.sh` rebuilds with no layer cache and clears the two build-artefact volumes before handing off to it (see "Major updates" in `docker/README.md`).
|
||||
|
||||
`CODEMAN_REPO_PATH` overrides which checkout is mounted. It defaults to the compose project's parent directory, so it normally needs no setting. Point it at a directory that is not a git checkout and in-app updates are reported as unavailable.
|
||||
|
||||
|
||||
@@ -10,15 +10,18 @@ this file covers only what the container changes.
|
||||
|
||||
## The short version
|
||||
|
||||
| Change in the release | Applied by |
|
||||
| -------------------------------- | ------------------------------------------------ |
|
||||
| Application code | The in-app updater |
|
||||
| `docker/server.Dockerfile` | `docker/Start-Codeman.sh` on the host |
|
||||
| `docker/docker-compose.yaml` | `docker/Start-Codeman.sh` on the host |
|
||||
| New key in `docker/.env.example` | Add it to `docker/.env`, then `Start-Codeman.sh` |
|
||||
| Change in the release | Applied by |
|
||||
| ----------------------------------------- | ------------------------------------------------------------------------------- |
|
||||
| Application code | The in-app updater |
|
||||
| `docker/server.Dockerfile` | `docker/Start-Codeman.sh` on the host |
|
||||
| `docker/docker-compose.yaml` | `docker/Start-Codeman.sh` on the host |
|
||||
| New key in `docker/.env.example` | Add it to `docker/.env`, then `Start-Codeman.sh` |
|
||||
| A major update, or a Node base-image bump | `docker/Update-Codeman.sh` on the host (no-cache rebuild + fresh build volumes) |
|
||||
|
||||
The in-app updater detects all three of the bottom rows itself and refuses with a
|
||||
The in-app updater detects the three middle rows itself and refuses with a
|
||||
message naming what changed, so you never have to work out which case you are in.
|
||||
`Update-Codeman.sh` is the heavier option for when `Start-Codeman.sh` is not
|
||||
enough: see "Major updates" in `docker/README.md`.
|
||||
|
||||
## Why the container needs its own path
|
||||
|
||||
@@ -224,7 +227,10 @@ the host and the in-app path works from then on.
|
||||
|
||||
**Resetting the build artefacts** — `docker compose down -v`, then
|
||||
`Start-Codeman.sh`. This discards the named volumes and re-seeds them from a fresh
|
||||
image.
|
||||
image. `docker/Update-Codeman.sh` scripts the same reset by default for the two
|
||||
build-artefact volumes (`codeman-node-modules`, `codeman-dist`) only, plus an
|
||||
unconditional `--no-cache` rebuild, which a plain `Start-Codeman.sh` run does not
|
||||
force on its own. See "Major updates" in `docker/README.md`.
|
||||
|
||||
## Disabling it
|
||||
|
||||
|
||||
@@ -0,0 +1,380 @@
|
||||
# Installer v2: three questions, then a URL you can open on your phone (Plan)
|
||||
|
||||
Status: **Phase 1 IMPLEMENTED (2026-09-20)**, phases 2 and 3 open. It builds on
|
||||
`docs/tailscale-installer-plan.md` (implemented 2026-08-04), which made Tailscale a
|
||||
guided option; this round makes it the thing the install ENDS on, and makes the whole
|
||||
installer shorter to sit through. Owner decisions taken before implementation: rename
|
||||
is opt-in and **defaults to no everywhere** (the machine name is used for other things);
|
||||
the URL keeps the node name unless asked; `codeman-<hostname>` is the suggested name;
|
||||
sub-path is the default for an occupied `:443`.
|
||||
|
||||
Verification record for phase 1 (all on the maintainer's box, 2026-09-20):
|
||||
|
||||
- `test/install-sh-invariants.test.ts` (28 tests, incl. the new Tailscale safety pins)
|
||||
and the detection-parity test pass; `bash -n` passes.
|
||||
- Every new decision function driven with stubbed tailscale state under **bash 5.2 and
|
||||
bash 3.2** (the `bash:3.2` container CI uses): flags, the launch default, the serve
|
||||
shape for free / ours / occupied `:443` (all four answers plus the non-interactive
|
||||
default), the three serve commands, the rename question (Enter keeps the name; `--yes`
|
||||
and non-interactive never rename; `codeman-*` nodes are skipped; `--name` is
|
||||
sanitized), `run_step` success/failure/stdin, the unit round-trip of
|
||||
`CODEMAN_BASE_URL`/`CODEMAN_PORT`/an escaped password, and the done screen.
|
||||
- A full non-interactive install into a sandboxed `HOME` with `CODEMAN_TAILSCALE=1`:
|
||||
preflight summary, kept the existing prod mapping (no serve mutation), clone 2 s,
|
||||
`npm install` 18 s, build 23 s, symlink, done screen; `install.sh status` on a pty
|
||||
renders the QR code. Nothing on the real system changed.
|
||||
- **Sub-path mode end to end over the real tailnet**: an isolated Codeman
|
||||
(`CODEMAN_INSTANCE`, port 3999, `--base-url /codeman`) behind
|
||||
`tailscale serve --https=8445 --set-path /codeman 3999` answered `/codeman/api/status`,
|
||||
`/codeman/` (with `<base href="/codeman/">` and `__CODEMAN_BASE__="/codeman"`), the
|
||||
hashed CSS/JS, `/codeman` without a slash, and the SSE stream; mapping and server
|
||||
removed afterwards. **Correction to section 2**: serve STRIPS the mount prefix
|
||||
before proxying (a direct `/codeman/api/status` on the server is 404 while the same
|
||||
path through serve is 200). That is fine because Codeman's ingress tolerates
|
||||
unprefixed requests; `--base-url` is needed for the URLs Codeman EMITS, not for
|
||||
what it receives.
|
||||
- Not yet exercised on a fresh machine (unchanged from the previous plan): Tailscale
|
||||
absent / logged out / HTTPS toggle off, the rename against a real node (the
|
||||
off-rename-re-add order is implemented but only unit-driven), macOS, uninstall. The
|
||||
Mac mini and a throwaway VM are the venues; see section 8.
|
||||
- **Review fixes (2026-09-21)**, from the two reviews on PR #460 (DeepSeek Harness, then
|
||||
Claude): the done screen's Start line is composed from every non-default value
|
||||
(`start_command_hint`, shared with the exec branch as `export_bind_env`), so "do not
|
||||
start" under a sub-path or a custom port no longer prints a bare `codeman web`; the
|
||||
`--lan`/`--tailscale`/env preset paths keep an existing password instead of rewriting
|
||||
the unit open; `--password`/`--port` flip `RECONFIGURE` so they reach the unit;
|
||||
`install.sh name` re-syncs the unit's base URL after a rename; the sudo keepalive is
|
||||
ended before the `exec` into the foreground server; Ctrl+C in the HTTPS-toggle poll
|
||||
skips Tailscale instead of killing the run; `uninstall` asks before removing a
|
||||
LaunchDaemon it never wrote; a foreign LaunchDaemon gets a restart hint and the done
|
||||
screen stops claiming the new build is running; the preflight summary reads the
|
||||
Tailscale state without node; the LAN security notice uses the configured port; a
|
||||
bare re-run ends on the done screen; a build failure after a rename names the
|
||||
`install.sh tailscale` recovery; `TS_JOINED_HERE` is gone.
|
||||
|
||||
Goal, in one sentence: a user runs the one-liner, answers at most three questions, walks
|
||||
away during the build, and comes back to `https://<name>.<tailnet>.ts.net` printed with a
|
||||
QR code, already answering, on every device in their tailnet. That is exactly the
|
||||
maintainer's own production setup (`tnode.tailf80371.ts.net` fronting `127.0.0.1:3000`),
|
||||
and the installer should produce it without the user knowing what `tailscale serve` is.
|
||||
|
||||
## 1. Where the installer is today
|
||||
|
||||
Facts from reading `install.sh` (2886 lines, 19 `prompt_yes_no` sites) and the live
|
||||
Tailscale state on the maintainer's box (tailscale 1.102.2, user-owned node, MagicDNS +
|
||||
HTTPS certs on, serve mapping `443 -> https+insecure://localhost:3000`).
|
||||
|
||||
**The order is backwards for a human.** The flow is: detect -> ask about git -> ask about
|
||||
node -> ask about tmux -> ask about build tools -> AI CLI menu -> ask about cloudflared ->
|
||||
clone -> `npm install` -> build (minutes) -> **then** the network-access question -> the
|
||||
Tailscale sub-steps (install? login URL, sudo for operator, admin-console toggle loop) ->
|
||||
the launch menu (no default; a bare Enter re-prompts) -> tunnel-service question. A fresh
|
||||
Ubuntu server taking the Tailscale route answers roughly ten prompts plus two to four sudo
|
||||
password prompts, split around a multi-minute build. The user cannot walk away at any
|
||||
point, and the question that matters most (how do I reach it) comes last.
|
||||
|
||||
**The Tailscale flow works but was never exercised on a fresh machine.** The previous
|
||||
plan's manual matrix still lists items 1-4, 7 and 10-12 (Tailscale absent, logged out,
|
||||
HTTPS toggle off, port 443 occupied, macOS, uninstall, phone PWA) as untested. The
|
||||
maintainer's own verification was the idempotent "kept as-is" path.
|
||||
|
||||
**The URL is the machine's name, full stop.** `setup_tailscale_serve` derives it from
|
||||
`.Self.DNSName`, and nothing lets the user influence it. A second Codeman on the same
|
||||
tailnet is `macminis-mac-mini.tailf80371.ts.net`, which tells you nothing about Codeman.
|
||||
|
||||
**Port 443 taken means give up or clobber.** If another app already owns the root of
|
||||
`:443`, the only offer is "replace it?" (default no), and declining falls back to
|
||||
local-only. Codeman already supports running under a sub-path (`--base-url`), and
|
||||
Tailscale serve supports mounting a path (`--set-path`), so there is a third answer nobody
|
||||
is offered.
|
||||
|
||||
**The result is invisible afterwards.** Once the terminal scrolls away, nothing in the app
|
||||
or the CLI tells the user their Tailscale URL again. `codeman doctor` does not probe
|
||||
Tailscale; App Settings -> Remote access shows only the Cloudflare tunnel.
|
||||
|
||||
**Two service writers exist.** `install.sh` carries its own plist/unit generator (~180
|
||||
lines) next to `codeman service install` (`src/service-installer.ts`). They agree on the
|
||||
job name by design, but the bash copy is the one that writes `CODEMAN_PASSWORD` into the
|
||||
unit, so they cannot simply be merged. Left as-is in this plan (see section 9).
|
||||
|
||||
## 2. What Tailscale makes possible for the name (researched 2026-09-20)
|
||||
|
||||
| Option | Resulting URL | What it needs | Side effects | Verdict |
|
||||
| ------ | ------------- | ------------- | ------------ | ------- |
|
||||
| **A. Node name** (today) | `https://tnode.tailf80371.ts.net` | `tailscale serve --bg 3000` | none | **Default.** Zero admin-console work, matches the maintainer's prod. |
|
||||
| **B. Rename the node** | `https://codeman-tnode.tailf80371.ts.net` | `tailscale set --hostname codeman-<host>` (operator or root) | Renames the machine tailnet-wide: ssh targets, other serve URLs, the admin console entry. Tailscale de-dups a clash as `-1`. The cert follows the new name. | **Opt-in, default NO everywhere** (owner decision 2026-09-20: the machine is used for other things, so a bare Enter never renames it). The proposal was YES when the installer itself had just joined the tailnet; rejected. |
|
||||
| **C. Tailscale Service** | `https://codeman.tailf80371.ts.net` | tailscale >= 1.86 on the host; the host must have a **tag-based identity** ("You cannot use a device authenticated with a user account as a Service host"); the service is defined in the admin console first; the host is then approved there (or via `autoApprovers.services`). Public beta since 2025-10-28, all plans. | Re-authenticating a personal machine as a tagged node changes its identity (SSH ACLs, user attribution). Known daemon quirk: approval is not picked up until `serve clear` + re-advertise (tailscale/tailscale#18821). | **Detect and hint only** in this round. The maintainer's own node has `Self.Tags: null`, so it could not host one without re-tagging. Worth a real flow once someone with a tagged fleet asks. |
|
||||
| **D. Sub-path** | `https://tnode.tailf80371.ts.net/codeman` | `tailscale serve --bg --set-path /codeman 3000` plus `--base-url /codeman` on the server | Codeman runs under a prefix. Hooks are unaffected (they hit the raw port with no prefix, which `rewriteUrl` already tolerates). Serve forwards the prefix unchanged, which is exactly the shape `--base-url` was built for. | **The answer when `:443` root is already taken.** Replaces today's replace-or-nothing prompt. |
|
||||
| **E. Second port** | `https://tnode.tailf80371.ts.net:8443` | `tailscale serve --bg --https=8443 3000` | Port in the URL; the beta-preview recipe already uses this. | Fallback when the user rejects D. |
|
||||
| Funnel (public internet) | `https://tnode.tailf80371.ts.net` from anywhere | `tailscale funnel` | Public exposure; different risk class. | **Out of scope**, as before. Docs only, with the password warning. |
|
||||
|
||||
Sources: Tailscale Services docs (`tailscale.com/docs/features/tailscale-services`), the
|
||||
Services beta announcement (`tailscale.com/blog/services-beta`), machine names
|
||||
(`tailscale.com/kb/1098/machine-names`), the serve CLI reference
|
||||
(`tailscale.com/docs/reference/tailscale-cli/serve`), the macOS variants page
|
||||
(`tailscale.com/docs/concepts/macos-variants`), and `tailscale serve --help` on 1.102.2
|
||||
(which lists `--service`, `--set-path`, `--yes`, `advertise`, `get-config`/`set-config`).
|
||||
|
||||
**Trap for option B (verify on the Mac mini before shipping):** the serve config is keyed
|
||||
by `host:port` using the DNS name at configuration time (`"Web": {"tnode.tailf80371.ts.net:443": ...}`
|
||||
in `serve status --json`). Renaming a node after serve is configured most likely orphans that
|
||||
entry: the handler lookup uses the current name and never matches the old key, and the only
|
||||
tool that removes a stale key is `serve reset`, which this installer must never run. So the
|
||||
order is **rename first, then configure serve** on a fresh install, and on a retrofit
|
||||
(`install.sh name`) **turn our mapping off, rename, wait for `.Self.DNSName` to change,
|
||||
re-add**.
|
||||
|
||||
## 3. Target UX
|
||||
|
||||
### 3.1 Three questions, then walk away
|
||||
|
||||
```
|
||||
Codeman installer
|
||||
|
||||
Found: git, Node 22.14, tmux 3.4, build tools Missing: nothing
|
||||
AI CLIs: Claude Code (~/.local/bin/claude)
|
||||
Tailscale: connected as tnode (tailf80371.ts.net)
|
||||
Existing: none
|
||||
|
||||
1/3 How should the dashboard be reachable?
|
||||
1) Tailscale https://tnode.tailf80371.ts.net (recommended, already connected)
|
||||
2) Any device on your network (0.0.0.0, password required)
|
||||
3) This machine only (127.0.0.1)
|
||||
Choose [1/2/3] (default 1):
|
||||
|
||||
2/3 Name this machine "codeman-tnode" on your tailnet? [y/N]
|
||||
(only shown for option 1; default no, always)
|
||||
|
||||
3/3 Run Codeman as a background service that starts on boot? [Y/n]
|
||||
|
||||
Installing… this takes a few minutes. You can leave this running.
|
||||
✓ dependencies ✓ clone ✓ build (2m 41s) ✓ service ✓ tailscale serve
|
||||
```
|
||||
|
||||
Rules that make this work:
|
||||
|
||||
- **Every step that needs a human runs BEFORE the build.** The dependency consent, the
|
||||
AI CLI menu, the Tailscale install consent, the `tailscale up` login URL, the operator
|
||||
grant, and the tailnet HTTPS toggle all move into the question phase. The build, the
|
||||
service, `tailscale serve` and the verification are unattended.
|
||||
- **One consent for all missing system packages.** "Install git, Node 22 and build tools
|
||||
now? [Y/n]" replaces four separate prompts. Each package still runs its own
|
||||
distro-specific installer.
|
||||
- **One sudo prompt.** When anything needs root (packages, the Tailscale installer,
|
||||
`tailscale up`, the operator grant), the installer says so once, runs `sudo -v`, and keeps
|
||||
the timestamp alive in a background loop until it exits. macOS needs no sudo for the
|
||||
Tailscale GUI-app CLI and the pattern still holds for Homebrew packages.
|
||||
- **Service is the default.** Enter on the last question installs the service; "run in
|
||||
this terminal" and "don't start" stay reachable by answering, and by flag.
|
||||
- **The cloudflared question is gone from the main flow.** It is optional, defaults to
|
||||
no, and has an in-app toggle (App Settings -> Remote access). The done screen mentions it
|
||||
only when `cloudflared` is already installed. The Linux tunnel-service prompt goes with it.
|
||||
- **The HTTPS-certificates toggle no longer asks "re-check now?"** The installer prints the
|
||||
admin URL, opens it in a browser when one is available (`xdg-open` / `open`, never on a
|
||||
headless box), and polls `tailscale status --json` every 5 s for up to 5 minutes. Ctrl+C or
|
||||
the timeout falls back exactly as today.
|
||||
- **Progress, not silence.** `npm install` and `npm run build` run behind one line each
|
||||
with elapsed time; their output goes to `~/.codeman/install.log` and is printed only on
|
||||
failure, with the exact retry command.
|
||||
|
||||
### 3.2 The done screen
|
||||
|
||||
One block, the URL first, a QR code the phone can scan, and nothing the user does not need
|
||||
right now.
|
||||
|
||||
```
|
||||
✓ Codeman 1.31.0 is running
|
||||
|
||||
Your tailnet: https://codeman-tnode.tailf80371.ts.net (HTTPS, any of your devices)
|
||||
This machine: http://localhost:3000
|
||||
|
||||
▄▄▄▄▄▄▄ ▄ ▄▄ ▄▄▄▄▄▄▄
|
||||
█ ▄▄▄ █ ▄▄▀ ▄ █ ▄▄▄ █ scan with your phone
|
||||
█ ███ █ ███▀▀ █ ███ █
|
||||
█▄▄▄▄▄█ █ ▄ █ █▄▄▄▄▄█
|
||||
|
||||
Manage systemctl --user restart codeman-web · journalctl --user -u codeman-web -f
|
||||
Update re-run the install line, or App Settings → System → Updates
|
||||
Docs https://github.com/Ark0N/Codeman/wiki
|
||||
|
||||
Security: Codeman binds 127.0.0.1. Tailscale authenticates every device before a
|
||||
packet reaches it. Details: docs/security-architecture.md
|
||||
```
|
||||
|
||||
The QR comes from the `qrcode` package Codeman already depends on
|
||||
(`node -e "require('qrcode').toString(url, {type:'terminal', small:true}, …)"` from
|
||||
`$INSTALL_DIR`, verified locally: 17 rows by 45 columns). Skipped when the terminal has no
|
||||
color support or fewer than 50 columns. The QR encodes the plain URL, not an auth token:
|
||||
the tailnet is the login.
|
||||
|
||||
### 3.3 Express mode and flags
|
||||
|
||||
Env vars stay (`CODEMAN_TAILSCALE=1`, `CODEMAN_HOST`, `CODEMAN_PASSWORD`,
|
||||
`CODEMAN_NONINTERACTIVE=1`, `CODEMAN_PORT`). Flags are added because they are
|
||||
discoverable from the one-liner and pipe through `bash -s --`:
|
||||
|
||||
```bash
|
||||
curl -fsSL https://getcodeman.com/install | bash -s -- --tailscale --service
|
||||
curl -fsSL https://getcodeman.com/install | bash -s -- --lan --password 'x' --service
|
||||
curl -fsSL https://getcodeman.com/install | bash -s -- --local --run
|
||||
curl -fsSL https://getcodeman.com/install | bash -s -- --tailscale --name codeman-build --yes
|
||||
```
|
||||
|
||||
| Flag | Meaning |
|
||||
| ---- | ------- |
|
||||
| `--tailscale` / `--lan` / `--local` | Answer 1/3 (same semantics as `CODEMAN_TAILSCALE=1`, `CODEMAN_HOST=0.0.0.0`, `CODEMAN_HOST=127.0.0.1`) |
|
||||
| `--name <n>` / `--no-rename` | Answer 2/3: rename the node to `<n>`, or never ask |
|
||||
| `--service` / `--run` / `--no-start` | Answer 3/3 |
|
||||
| `--yes` | Accept every default, still prompt for a login URL (a human must open it) |
|
||||
| `--password <p>` | Same as `CODEMAN_PASSWORD` |
|
||||
| `--port <n>` | Same as `CODEMAN_PORT`; the serve target follows it |
|
||||
|
||||
`--yes` differs from `CODEMAN_NONINTERACTIVE=1`: it is the interactive user saying "I trust
|
||||
the defaults", so it may install software and may wait on a login URL. Non-interactive stays
|
||||
the CI contract and never installs Tailscale.
|
||||
|
||||
## 4. The Tailscale flow, v2
|
||||
|
||||
The state machine from the previous plan stays; these are the changes.
|
||||
|
||||
1. **Preflight, before the build** (`tailscale_preflight`): installed? -> install
|
||||
(Linux: official script; macOS: brew cask, else download link and wait). Logged in? ->
|
||||
`tailscale up` with the URL printed prominently and a 5-minute poll. Operator (Linux):
|
||||
grant once under the single sudo session. HTTPS certs: poll instead of ask (Ctrl+C
|
||||
during the poll skips Tailscale for this run rather than ending the installer). The
|
||||
rename default does not depend on whether this run performed the login (decided NO
|
||||
everywhere), so nothing records it.
|
||||
2. **Name** (`tailscale_choose_name`, question 2/3): shown only on the Tailscale route.
|
||||
Default `codeman-<oshostname>` sanitized to `[a-z0-9-]`, max 63. Applied with
|
||||
`ts_cmd_serve set --hostname`, then poll `.Self.DNSName` until it carries the new name
|
||||
(up to 60 s). Order matters: this runs before any serve mutation (section 2 trap).
|
||||
Declining keeps the node name. On a re-run against a node already named `codeman-*`,
|
||||
the question is skipped.
|
||||
3. **Serve, after the service is up** (`setup_tailscale_serve`): unchanged idempotent
|
||||
"kept as-is" path first. When `:443` root belongs to another target, the new prompt is:
|
||||
|
||||
```
|
||||
tailscale serve already sends https://tnode.tailf80371.ts.net to port 8080.
|
||||
1) Add Codeman under a path: https://tnode.tailf80371.ts.net/codeman (default)
|
||||
2) Use another port: https://tnode.tailf80371.ts.net:8443
|
||||
3) Replace the existing mapping with Codeman
|
||||
4) Skip Tailscale for now
|
||||
```
|
||||
|
||||
Option 1 writes `--base-url /codeman` into the service unit (it is a `WebLaunchOptions`
|
||||
field already, and `buildWebArgs` carries it) and runs
|
||||
`tailscale serve --bg --set-path /codeman <port>`. Option 2 runs `--https=8443`.
|
||||
`detect_tailscale_serve_url` learns to recognize all three shapes (root, path, port) so
|
||||
uninstall, the security notice and the re-run default keep working.
|
||||
4. **Warm the certificate.** Right after serve is configured, fire one background
|
||||
`curl -sk https://<url>/api/status` so Let's Encrypt issuance overlaps the rest of the
|
||||
install instead of adding 30 s to the verify step.
|
||||
5. **Verify** as today (200 or 401 on `/api/status`), with the path-aware URL.
|
||||
6. **Services hint** (option C): when `.Self.Tags` is non-empty and `serve --help`
|
||||
lists `--service`, the done screen adds one line: "This is a tagged node, so it can also
|
||||
host `https://codeman.<tailnet>.ts.net` as a Tailscale Service: see Remote Access in the
|
||||
wiki." No flow, no prompt.
|
||||
7. **macOS**: the App Store and Standalone variants cannot run before login, so a
|
||||
LaunchAgent plus serve only comes back after someone logs in. The done screen says so on
|
||||
macOS. The Mac mini (`arbbot`, headless, system LaunchDaemon) is the reference for the
|
||||
"headless Mac" caveat, and `install.sh` must keep refusing to replace a LaunchDaemon it
|
||||
did not write (today it removes one; that is a bug for the Mac mini and is fixed here:
|
||||
detect `UserName` in the daemon plist and leave it alone with a message).
|
||||
8. **Uninstall** additionally offers to restore the original node name when this installer
|
||||
renamed it (the original is recorded in `~/.codeman/install.json`, the one marker file
|
||||
this feature adds, because tailscaled does not remember previous names).
|
||||
9. **Subcommands**: `install.sh tailscale` (unchanged purpose, now runs the v2 flow),
|
||||
`install.sh name [<n>]` (rename with the off/rename/re-add dance), `install.sh status`
|
||||
(prints the done screen again, URL and QR included, for the "what was my URL" moment).
|
||||
|
||||
## 5. In-app: the URL stays discoverable
|
||||
|
||||
Small, read-only, and the first server-side code this feature has ever needed.
|
||||
|
||||
- **`GET /api/system/remote-access`** returns
|
||||
`{ tailscale: { installed, connected, dnsName, url, mode: 'root'|'path'|'port'|null } }`
|
||||
by running `tailscale status --json` and `tailscale serve status --json` through
|
||||
`execFile` with the existing exec timeout, cached 30 s, resolved through the same
|
||||
`get_tailscale_path` search as the installer (PATH, then the macOS app bundle), and a
|
||||
no-op under `VITEST` like every other IO probe. Never mutates serve config.
|
||||
- **App Settings -> Remote access** gains a **Tailscale** row above the Cloudflare toggle:
|
||||
the URL as a copy chip, a QR button reusing `showTunnelQR`'s modal, and when nothing is
|
||||
configured a one-line hint with `bash ~/.codeman/app/install.sh tailscale`. The welcome
|
||||
screen's "open on your phone" affordance shows the same QR.
|
||||
- **`codeman doctor`** grows a `tailscale` entry under `other` in
|
||||
`config/dependency-registry.ts`: installed, connected, serving Codeman (URL). Pure
|
||||
engine, injectable probe host, like the existing rows.
|
||||
- No new SSE event, no settings key, no state.json change.
|
||||
|
||||
## 6. Security posture
|
||||
|
||||
Nothing widens. The bind stays loopback; the tailnet is the authentication boundary;
|
||||
`.ts.net` is already in `DEFAULT_TRUSTED_HOST_SUFFIXES`. New surfaces are read-only
|
||||
probes. `install.sh` still never runs `tailscale serve reset`, still touches only the
|
||||
mapping it created, and gains one more never: it never advertises a Tailscale Service or
|
||||
runs `tailscale funnel`. The sudo keep-alive loop is killed by the existing `cleanup` trap.
|
||||
The rename records the previous name locally and offers the reversal at uninstall.
|
||||
|
||||
## 7. Implementation inventory
|
||||
|
||||
| File | Change |
|
||||
| ---- | ------ |
|
||||
| `install.sh` | New `parse_flags`, `preflight_summary`, `ask_everything` (the three questions), `sudo_session`, `run_step` (spinner + log), `tailscale_preflight`, `tailscale_choose_name`, `tailscale_rename_node`, `print_done_screen`, `print_qr`, `status` subcommand, `name` subcommand. Modified: `main` (reordered into ask -> work -> done), `choose_network_binding` (question 1/3, same defaults), `setup_tailscale_serve` (path/port options), `detect_tailscale_serve_url` (three shapes), `setup_systemd_service`/`setup_launchd_service` (`--base-url`, LaunchDaemon guard), `uninstall` (rename reversal), header docs (flags). Removed from the main flow: the cloudflared prompt, the tunnel-service prompt. bash 3.2 rules unchanged. |
|
||||
| `src/web/routes/system-routes.ts` | `GET /api/system/remote-access` |
|
||||
| `src/tailscale-status.ts` (new) | Pure parser for the two JSON shapes + the IO wrapper; unit-tested against captured `serve status --json` fixtures (root, path, port, foreign target, none) |
|
||||
| `src/config/dependency-registry.ts`, `src/utils/dependency-checker.ts` | `tailscale` doctor row |
|
||||
| `src/web/public/index.html`, `settings-ui.js`, `panels-ui.js` | Tailscale row + QR, welcome-screen QR |
|
||||
| `test/install-sh-invariants.test.ts` | Extend: flags documented in the header, no `serve reset`, no `funnel`, no `--service` advertise, every serve mutation goes through `ts_cmd_serve`, rename happens before serve in `main` (static order check) |
|
||||
| `.github/workflows/ci.yml` | The bash 3.2 step additionally sources the script with stubbed `ts_cmd`/`ts_cmd_serve`/`read_reply` and drives `ask_everything` through all three answers and the 443-occupied menu |
|
||||
| `test/tailscale-status.test.ts`, `test/routes/system-routes-remote-access.test.ts` | Parser + route |
|
||||
| Docs | README install + remote-access sections, `docs/wiki/Installation.md`, `Remote-Access.md` (naming options table, Services caveat, path/port variants), `Mobile-Guide.md`, `Running-As-A-Service.md` (macOS login caveat), `FAQ.md`, `docs/security-architecture.md` §A, CLAUDE.md Scripts & Tunnel paragraph, `docs/tailscale-installer-plan.md` gets a pointer here. getcodeman.com copy lives outside the repo (maintainer handbook). |
|
||||
|
||||
Changeset: `minor` (new flags, new subcommands, new API route).
|
||||
|
||||
## 8. Test plan
|
||||
|
||||
Automated (the gate): the static invariants above, the bash 3.2 container drive of the
|
||||
question phase, the JSON parser fixtures, the route test.
|
||||
|
||||
Manual matrix, on a fresh Ubuntu 24 VM and on the Mac mini, since the previous plan's
|
||||
items never ran on a fresh machine:
|
||||
|
||||
1. Tailscale absent, declined -> local-only, done screen shows the retrofit command.
|
||||
2. Tailscale absent, accepted -> install, login URL, operator, certs toggle polled, rename
|
||||
question shown (default no), service, serve, URL verified, QR scans on a phone, PWA installs.
|
||||
3. Tailscale present and logged in on a pre-existing node -> rename default NO, URL is the
|
||||
node name, `serve status` gains exactly one entry.
|
||||
4. `:443` root occupied -> path option -> `https://<node>/codeman` answers, hooks still
|
||||
fire (raw port), `install.sh status` prints the path URL.
|
||||
5. Rename on a node that already has our serve mapping (`install.sh name`) -> off, rename,
|
||||
re-add, `serve status` has no stale key.
|
||||
6. Re-run the one-liner -> quiet update, binding and name preserved, no prompts.
|
||||
7. `--yes` end to end; `CODEMAN_NONINTERACTIVE=1` end to end (no software installed).
|
||||
8. Uninstall -> mapping removed, other mappings intact, rename reversal offered.
|
||||
9. Mac mini: LaunchDaemon left alone with the message; done screen carries the login caveat.
|
||||
|
||||
## 9. Phasing and open decisions
|
||||
|
||||
**Phase 1 (this round):** the reorder, the three questions, one consent + one sudo, flags,
|
||||
the done screen with QR, Tailscale preflight-before-build, the path/port answer for an
|
||||
occupied 443, the rename step, `status` and `name` subcommands, docs.
|
||||
|
||||
**Phase 2:** the in-app Tailscale row + QR, `codeman doctor` row, the `remote-access`
|
||||
route. Independent of phase 1 and useful on its own for existing installs.
|
||||
|
||||
**Phase 3 (optional):** replace the bash service writers with `codeman service install`
|
||||
once that command can carry `CODEMAN_PASSWORD` behind an explicit flag; and a Tailscale
|
||||
Services flow if a tagged-fleet user asks for `codeman.<tailnet>.ts.net`.
|
||||
|
||||
Decisions for the maintainer:
|
||||
|
||||
1. **Rename default.** Decided 2026-09-20: always NO; the yes answer, `--name` and
|
||||
`install.sh name` are the ways in. (The proposal was YES only when this run had joined
|
||||
the tailnet, NO otherwise; rejected because the host is used for other things.)
|
||||
2. **Name pattern.** `codeman-<hostname>` (proposed; unique per machine, and two Codemans
|
||||
on one tailnet stay distinguishable) versus plain `codeman` (nicer once, collides on the
|
||||
second install, Tailscale silently appends `-1`).
|
||||
3. **Path versus port** as the default answer for an occupied 443. Proposed: path, because
|
||||
the URL has no port and `--base-url` already exists for exactly this proxy shape.
|
||||
4. **Whether Phase 2 ships in the same release.** It is the part that helps people who
|
||||
installed months ago.
|
||||
@@ -66,6 +66,31 @@ each `(clientId, seq)` at most once, so a resend can't type the prompt twice.
|
||||
(the 200 is the client's ACK). `curl`/legacy callers omit the fields and always
|
||||
apply.
|
||||
|
||||
## Oversized input (issue #484)
|
||||
|
||||
Delivery has a third outcome besides "applied" and "retry": **refused for good**.
|
||||
Both transports refuse a frame longer than `MAX_INPUT_LENGTH` (64 KiB,
|
||||
`src/config/terminal-limits.ts`; the POST schema uses the same constant). Before
|
||||
#484 the client treated that like a transient failure, so an oversized paste sat
|
||||
at the head of the queue, was re-sent every 2 s forever, blocked every later
|
||||
input for the session, and came back from localStorage on each reload.
|
||||
|
||||
- `_sendInputAsync()` splits a paste over the frame limit into in-limit frames
|
||||
(`CodemanInputLimit.split`, constants.js, never cutting a surrogate pair). They
|
||||
go out in seq order, so the PTY sees one contiguous stream. A paste over
|
||||
`PASTE_MAX_CHARS` (1 MiB), or an oversized `useMux` write (line-oriented, never
|
||||
split), is refused with a toast and never queued.
|
||||
- The WebSocket answers an oversized sequenced frame with
|
||||
`{t:'ia', seq, err:'too_large', max}`; the client drops it with a toast. A
|
||||
client that predates `err` reads it as a plain ACK and drops it too.
|
||||
- The POST drain drops a frame answered `400`/`413` (`401`/`403` stay transient:
|
||||
an expired login delivers once the user signs in again).
|
||||
- `_loadReliableState()` prunes persisted frames over the limit, so a queue
|
||||
poisoned by an older build heals on the first load after upgrading.
|
||||
- ⚠️ The frontend limit (`INPUT_FRAME_MAX_CHARS`) and the composer's
|
||||
`COMPOSER_INPUT_FRAME_LIMIT` must equal `MAX_INPUT_LENGTH`; pinned by
|
||||
`test/input-size-limit.test.ts`.
|
||||
|
||||
## Known limitation
|
||||
|
||||
Dedup state is in-memory on the server. A **server restart** between a write and
|
||||
@@ -79,3 +104,5 @@ across the narrow restart window.
|
||||
semantics (monotonic, per-client, gap-tolerant, eviction-safe).
|
||||
- `test/routes/session-routes.test.ts` — POST `/input` applies a tagged
|
||||
`(clientId, seq)` once on redelivery; untagged input always applies.
|
||||
- `test/input-size-limit.test.ts`: one input limit on both sides, frame
|
||||
splitting, and dropping (never retrying) a frame refused for good (#484).
|
||||
|
||||
@@ -352,6 +352,177 @@ path but the SESSION (`session.remote`): a remote session never falls back to lo
|
||||
`fs`, and a local session never opens an ssh connection — including for attachment
|
||||
records, which are keyed to the session that registered them.
|
||||
|
||||
## Wake-on-LAN from user input
|
||||
|
||||
A durable remote session survives an SSH drop (COD-104/108), but nothing brought the
|
||||
HOST back. When the remote machine suspended, the local pane's `ssh` child **stalled**
|
||||
rather than exited: `tmux send-keys` SUCCEEDS against a stalled pane, so typed input
|
||||
vanished with no error anywhere, and without a keepalive the pane could look alive for
|
||||
the OS TCP timeout. The only recovery was waiting for the reconnect watcher, which
|
||||
gave up after ~13 minutes and, once exhausted, never retried.
|
||||
|
||||
An **optional** `wakeMac` (one or more MAC addresses, comma-separated) or `wakeCommand` on a
|
||||
remote host closes that: on user input, `POST /api/sessions/:id/input` probes the host, and if
|
||||
it is unreachable it wakes it, polls until the host answers, reattaches the pane
|
||||
(`Session.reattachRemote()`, which idempotently attaches the still-running remote tmux — the
|
||||
agent conversation is not restarted), and flushes the input that arrived meanwhile.
|
||||
Implementation: `src/remote-wake.ts`.
|
||||
|
||||
The same wake path also serves **opening** a session, which is where a sleeping host used to
|
||||
be a dead end: pressing Run on a remote case (`POST /api/quick-start`) or Attach on a
|
||||
discovered remote tmux session (`POST /api/sessions` + `attachRemoteSession`) probes the host
|
||||
first, and on a sleeping one wakes it, waits for SSH and only then runs the tmux prereq probe.
|
||||
Without that the run failed with `could not verify tmux on remote host …` — an ssh error that
|
||||
blames tmux for a machine that is merely suspended. The wait is **blocking** (the caller gets
|
||||
the session or the error) but bounded by `REMOTE_WAKE_REQUEST_READY_TIMEOUT_MS` (40 s) rather
|
||||
than the 90 s session default, because the dashboard sits behind a reverse proxy whose default
|
||||
`proxy_read_timeout` is 60 s: a longer wait would be cut off at the proxy while the session was
|
||||
still being created. The budget covers the whole request, not just the wait (40 s wake + 1.5 s
|
||||
probe + the tmux prereq probe's own 15 s timeout = 56.5 s worst case). A host with no wake target is not even probed on this path, so nothing
|
||||
changes for it, and `remote:hostWaking` is broadcast without a `sessionId` (the toast then reads
|
||||
"the session starts when it is back" — there is no session yet, and no input queued behind it).
|
||||
|
||||
Two wake paths, `wakeCommand` first because it is the explicit override:
|
||||
|
||||
- **`wakeMac`** — Codeman builds the magic packet itself (`buildMagicPacket`, six `0xFF`
|
||||
bytes then the MAC repeated 16×; the shape is asserted byte-for-byte) and broadcasts it
|
||||
over UDP port 9 (`sendWakePackets`). This is the normal case: no external script, and one
|
||||
MAC list per host instead of one per consumer.
|
||||
- **`wakeCommand`** — a single executable path, run WITHOUT a shell. For hosts that need a
|
||||
router/another machine to send the packet.
|
||||
|
||||
**UI**: a banner (`#hostWakeBanner`, `host-wake-ui.js`) appears while the ACTIVE remote
|
||||
session's host is unreachable — amber, since the Codeman session is healthy and only the
|
||||
machine is asleep. With a wake target the action is **Wake** (`POST /api/sessions/:id/wake`);
|
||||
with none it is **Configure WoL** and opens `#wakeConfigModal`, a small form for that host's
|
||||
`wakeMac`/`wakeCommand` that saves with `PUT /api/remote-hosts/:id` (in multi-user mode that
|
||||
GET is admin-only, so a non-admin is told the setting is admin-only instead of "host not
|
||||
found"). Reachability for the banner comes from `GET /api/sessions/:id/reachability`: once
|
||||
when the remote tab is activated (a user action), and every 30 s while the tab is visible
|
||||
**only for a host with a wake target** — each poll is a TCP connect to the host, and a timer
|
||||
that connects to a host Codeman could not wake anyway is exactly the timer-driven traffic
|
||||
the keepalive rule below rejects (it cannot wake a host, but it can keep an activity-based
|
||||
suspend timer from firing). A host the probe cannot reach (see the next section) is never
|
||||
polled. ⚠️ The button is pressed from the SAME
|
||||
dashboard as Run/Attach, so it holds its request open under the same proxy and uses the same
|
||||
40 s budget — and it **queues nothing**: browser keystrokes travel over the WebSocket, which
|
||||
deliberately does not pass through the registry (that is the hot path this feature keeps its
|
||||
hands off), so the banner says "waiting for the host to come back" for the button and only
|
||||
claims "input is queued" when the HTTP input path actually buffered bytes
|
||||
(`queuedInput` on the two SSE events).
|
||||
|
||||
**Hosts behind a jump host or SOCKS proxy are reachability-UNKNOWN.** The probe is a bare
|
||||
TCP connect to `host:port`, and a host reached through `jumpHost`, `socksProxy` or a
|
||||
`ProxyCommand`/`ProxyJump` in `extraSshOptions` does not answer that even while ssh works —
|
||||
the direct address may not route at all (the cloudflared case). Acting on the resulting
|
||||
"unreachable" verdict was wrong three times over: a permanent banner over a healthy session,
|
||||
a create-path error that replaced a genuine "needs tmux" with "not reachable", and — with a
|
||||
wake target configured — every HTTP input buffered for the life of the session, because the
|
||||
readiness poll could never succeed. `isProbeable()` (`remote-wake.ts`) decides from the
|
||||
proxy fields, which travel on `WakeableRemote`; for such a host the registry delivers input
|
||||
unchanged, `GET …/reachability` answers `reachable: null, probeable: false` (unknown is not
|
||||
`false`, and only a proven `false` raises the banner), the create/attach path is not gated
|
||||
(`ensureHostAwake` → `'unprobeable'`, handled like `'no-target'`), and the quick-start
|
||||
"not reachable" message is reserved for a **proven** unreachable host (`=== false`). A wake
|
||||
target can still be fired for it through `POST /api/sessions/:id/wake`, blind: the packet or
|
||||
command goes out and the response says only whether it did — no readiness poll, no reattach
|
||||
(the COD-108 watcher owns the pane once ssh works again), no "waking" toast.
|
||||
|
||||
The invariants worth keeping:
|
||||
|
||||
- **Authorization comes before the wake.** In multi-user mode the attach path
|
||||
(`POST /api/sessions` + `attachRemoteSession`) answers `403` to a non-admin BEFORE the
|
||||
host is looked up or probed: remote hosts are admin-only infrastructure everywhere else
|
||||
(the list is `[]` for a non-admin, write and discovery routes are `adminOnly`), and the
|
||||
wake spawns the host's `wakeCommand` or broadcasts a packet — a gate that came after the
|
||||
wake handed an unprivileged account a way to run that executable for any configured
|
||||
`hostId`, hold the request for the wake budget, and only then be refused for the
|
||||
workingDir. The quick-start path resolves its remote case through `canAccessOwned`
|
||||
first. Pinned in `test/routes/session-remote-wake.test.ts` (wake spy stays empty).
|
||||
- **The caller is told what happened to its bytes.** The non-wait input route answers
|
||||
`{buffered:true}` when the registry took the chunk and `{buffered:true, dropped:true}`
|
||||
when it was over the cap and is gone; the send-and-wait route answers `OPERATION_FAILED`
|
||||
when the host never comes back, like the create and attach paths, instead of writing
|
||||
into the stalled pane and reporting `delivered:true` plus a timeout. Flushed chunks are
|
||||
written with `fromUser`, so a first prompt that was buffered through a wake can still
|
||||
name the tab.
|
||||
- **Only an EXPLICIT request may wake a host:** user input on an established session, the wake
|
||||
button, or the user's own session create/attach request (`ensureHostAwake`). Everything that
|
||||
runs on a TIMER must never wake one — the COD-108 watcher, the server's dropped-session
|
||||
handler, boot recovery and session discovery have no access to the wake registry, and neither
|
||||
has the shared session service, because `cron-service.ts` builds sessions there with nobody
|
||||
waiting on the answer; a wake on such a path would re-wake the host seconds after every
|
||||
suspend, so it could never stay asleep (the same failure `hufflepuff-mcp-lazy` exists to
|
||||
prevent for MCP keepalives). A reachability check, a discovery listing and the tmux prereq
|
||||
probe never wake: they are questions, not actions. All of it is enforced by tests in
|
||||
`test/remote-wake.test.ts` (two wiring guards: one pins the importers — the route module and
|
||||
`server.ts`, which holds the registry for its LIFETIME only, `drop()` on session cleanup and
|
||||
`stop()` on shutdown — and one asserts `server.ts` calls nothing but those two, while
|
||||
`ensureHostAwake` has exactly one caller file) and `test/routes/session-remote-wake.test.ts`,
|
||||
not by comments.
|
||||
- **Detection is a bare TCP connect** to the SSH port (then the configured `port`, else 22),
|
||||
throttled per session, and only for wake-enabled hosts. No `ServerAliveInterval` is added to
|
||||
the launch command: keepalives push bytes into an otherwise idle connection every interval,
|
||||
which is exactly what a byte-threshold idle detector must not count as activity. A probe is
|
||||
~200 bytes per 30 s, orders of magnitude below any such threshold, and the SYN alone cannot
|
||||
wake a host.
|
||||
- **Input is buffered while a wake is in flight** (`REMOTE_WAKE_PENDING_MAX_BYTES`,
|
||||
oldest whole chunks dropped, bounded so user input cannot grow memory) and flushed in
|
||||
order after the reattach, with a settle delay so bytes cannot land in a still-connecting
|
||||
pane. ⚠️ A chunk LARGER than the cap (one big paste is one `input` value) is dropped
|
||||
**outright**, never trimmed: it was never typed character by character, so its tail is not
|
||||
"what the user just typed" but a fragment of a command they never sent — the drop is logged
|
||||
instead. ⚠️ Only the HTTP input route reaches the registry; the **WebSocket keystroke path
|
||||
is deliberately NOT wake-aware**, so typing into a sleeping host sends nothing and queues
|
||||
nothing (the banner's Wake button is the recovery for that case, which is why it must not
|
||||
promise queued input). The **send-and-wait** path blocks on the wake instead — its response
|
||||
is open anyway, and buffering would break the wait contract. ⚠️ A flush write that FAILS
|
||||
drops the whole remaining buffer (logged) rather than retaining it: the wake still resolves
|
||||
and marks the host reachable, so the next input takes the deliver path while a retained
|
||||
chunk would wait for the NEXT wake — replayed hours later, after everything typed since,
|
||||
possibly ending in a carriage return. Same policy as the oversized paste.
|
||||
- **The command runs without a shell** (`spawn(path, [], { stdio: 'ignore' })` — `shell`
|
||||
defaults to `false`), the schema
|
||||
requires a single executable path (no arguments, no `$`/backtick), and `wakeMac` is a
|
||||
structural hex-pair allowlist. A broken or missing wake target fails the wake, never the
|
||||
input route.
|
||||
- **`wakeMac`/`wakeCommand` are host-level config, refreshed on recovery AND live**
|
||||
(`rehydrateRemoteHostFields` in `src/remote-hosts.ts` plus `RemoteWakeDeps.resolveRemote`).
|
||||
A session's `remote` block is persisted at launch time, so a field added to
|
||||
`remote-hosts.json` later would otherwise never reach an already-running session — not even
|
||||
across a Codeman restart, and certainly not right after saving the banner's config dialog.
|
||||
Recovery rehydration covers restarts, the (throttled, cache-backed) resolver covers the live
|
||||
session; the host config is authoritative for both (removing the field disables the feature
|
||||
again). Other host-level fields deliberately stay as persisted, so neither path can
|
||||
silently re-point an existing pane's SSH options.
|
||||
- **UI/SSE**: `remote:hostWaking` and `remote:hostWakeFailed` (plus the reused
|
||||
`remote:sessionReconnected`) drive the banner and toasts, all from `host-wake-ui.js` —
|
||||
its handlers are the ONLY definitions, since a second one in another mixin would be
|
||||
silently shadowed by script order. Both carry `queuedInput`, which is true only when the
|
||||
server actually holds bytes for that session — the wording keys off that, not off "a wake
|
||||
is running", so the button path never claims input is queued. In multi-user mode the
|
||||
whole `remote:` family is **session-scoped** (`deriveSseHint`, `server.ts`): an event with
|
||||
a `sessionId` reaches that session's owner, and the create/attach wake — which has no
|
||||
session yet — carries the requesting `username` instead (`ensureHostAwake({ requestedBy })`),
|
||||
since its payload names a `hostId`/`label` that `GET /api/remote-hosts` withholds from
|
||||
non-admins. With neither, it reaches admins only.
|
||||
- **No real IO under vitest.** `probeRemoteHostReachable`, `runRemoteWakeCommand` and the
|
||||
default UDP socket of `sendWakePackets` throw under `VITEST` (as `remote-files.ts` does),
|
||||
so a test that reaches the defaults fails loudly instead of connecting, spawning or
|
||||
broadcasting from CI. Every consumer injects its IO (`RemoteWakeDeps`, the socket
|
||||
factory); `createDefaultRemoteWakeDeps({ probe })` also polls readiness with THAT probe,
|
||||
which is the leak the guard found.
|
||||
|
||||
Tests: `test/remote-wake.test.ts` (decision/throttle table, single-flight registry,
|
||||
buffering + flush order, MAC parsing/magic packet, live host-config resolution, the proxied
|
||||
host, SSE payload routing, the vitest IO guard, and the wiring guard),
|
||||
`test/routes/session-remote-wake.test.ts` (the input route buffers instead of writing into a
|
||||
sleeping host — and writes straight into a proxied one —, the reachability route never wakes
|
||||
and reports a proxied host as unknown, and the wake route reports the no-target case the UI
|
||||
turns into "configure WoL"), `test/sse-routing-remote.test.ts` (multi-user routing of the
|
||||
`remote:` family) and `test/host-wake-banner.test.ts` (banner visibility and when the poller
|
||||
may connect).
|
||||
|
||||
## API
|
||||
|
||||
Routes are registered in `src/web/routes/case-routes.ts`:
|
||||
@@ -365,6 +536,10 @@ Routes are registered in `src/web/routes/case-routes.ts`:
|
||||
| `GET` | `/api/remote-hosts/:hostId/sessions` | Discover `codeman-*` sessions on the host (COD-105; `listRemoteCodemanSessions`, never errors) |
|
||||
| `POST` | `/api/cases/remote-link` | Link a case to a remote host (creates the `RemoteCase`) |
|
||||
|
||||
`RemoteHost` accepts the optional `wakeMac` (magic packet, sent by Codeman) and `wakeCommand`
|
||||
(single executable path, run without a shell, takes precedence) — see **Wake-on-LAN from user
|
||||
input** above.
|
||||
|
||||
Attaching to a discovered session is a **session-create** path, not a host route:
|
||||
`POST /api/sessions` accepts `attachRemoteSession: { hostId, remoteSessionName }`
|
||||
(schema in `schemas.ts`; `remoteSessionName` must match `^codeman-[a-zA-Z0-9._-]+$`),
|
||||
|
||||
@@ -270,9 +270,15 @@ tailscale serve --bg 3000 # HTTPS at https://<node>.<tailnet>.ts.net
|
||||
Only devices on your tailnet can reach it; Tailscale handles identity and
|
||||
terminates TLS with a real Let's Encrypt certificate (so PWA install and web
|
||||
push work). No app password and no `0.0.0.0` bind required. (This is the
|
||||
maintainer's production setup.) `CODEMAN_TAILSCALE=1` presets the choice for
|
||||
automation; the installer never runs `tailscale serve reset` and never touches
|
||||
serve mappings other than `443 -> Codeman's port`.
|
||||
maintainer's production setup.) `CODEMAN_TAILSCALE=1` or `--tailscale` presets
|
||||
the choice for automation. When `:443` on the node already belongs to another
|
||||
app, the installer mounts Codeman under `/codeman` (`tailscale serve --set-path`
|
||||
plus `--base-url`, which keeps the loopback bind and the same host guard) or on a
|
||||
second port rather than replacing it. The installer never runs `tailscale serve
|
||||
reset`, never touches serve mappings other than the one it created, never opens a
|
||||
`tailscale funnel` (public internet, a different risk class) and never advertises
|
||||
a Tailscale Service. Renaming the node (`--name`, `install.sh name`) is opt-in
|
||||
and defaults to no, because the tailnet name is also the machine's SSH identity.
|
||||
|
||||
### B. Authenticated cloudflared tunnel + password
|
||||
|
||||
@@ -491,7 +497,7 @@ production layout (`~/.codeman`, `-L codeman`, port 3000).
|
||||
Docker cases (1.4.0) run a session inside a per‑case container instead of on the host. The security posture:
|
||||
|
||||
- **Hardened create flags, always** — `--cap-drop ALL`, `--security-opt no-new-privileges`, `--pids-limit` (fork‑bomb guard), `--memory` == `--memory-swap` (a real OOM cap), `--init`, and non‑root: `--user <hostUid>:0` on Linux (host uid → workspace files stay host‑owned; GID 0 keeps `$HOME` writable), `--userns=keep-id` on rootless Podman. **Never** `--privileged`, and **never** the docker socket — the pure builder in `docker-hosts.ts` cannot emit them and the schema cannot represent them.
|
||||
- **Credentials never enter an image** — the convenient default bind‑mounts host cred dirs (`~/.claude`, `~/.codex`, `~/.gemini` — which also carries Antigravity's `antigravity-cli/` state — `~/.config/{gcloud,opencode}`, five seeded files from `~/.pi/agent`, and three from `~/.grok`) read‑write. Bind mounts are physically excluded from `docker commit`, so exported images are secret‑free. API‑key CLIs get their key as an exec‑time NAME‑ONLY `--env OPENAI_API_KEY` (no `=value`, no `ps` leak, never committed); a create‑time `-e` for a secret is never used. The **sealed** profile (`mountCredentials:false` + `network:none`) drops the host mounts; full‑image export is then refused (an in‑container login would ride the committed layer) unless a pre‑commit scrub is opted into.
|
||||
- **Credentials never enter an image** — the convenient default bind‑mounts host cred dirs (`~/.claude`, `~/.codex`, `~/.gemini` — which also carries Antigravity's `antigravity-cli/` state — `~/.config/{gcloud,opencode}`, five seeded files from `~/.pi/agent`, three from `~/.grok`, and, only when their opt-in switches `CODEMAN_AGENT_IMAGE_INSTALL_GH` / `_AZ` are `1`, `~/.config/gh/{hosts.yml,config.yml}` and the sign-in files from `~/.azure`) read‑write. Bind mounts are physically excluded from `docker commit`, so exported images are secret‑free. API‑key CLIs get their key as an exec‑time NAME‑ONLY `--env OPENAI_API_KEY` (no `=value`, no `ps` leak, never committed); a create‑time `-e` for a secret is never used. The **sealed** profile (`mountCredentials:false` + `network:none`) drops the host mounts; full‑image export is then refused (an in‑container login would ride the committed layer) unless a pre‑commit scrub is opted into.
|
||||
- **Blast radius — accept it explicitly** — the convenient profile mounts an arbitrary host workspace RW plus the host credential dirs RW into a network‑enabled container, so container‑run agent code can read/modify those host trees and reach the network at once. Still a net improvement over today's on‑host `--dangerously-skip-permissions` execution; use the sealed profile for genuinely untrusted work.
|
||||
- **Import is untrusted‑bundle‑safe** — `/api/docker-cases/import` validates the manifest + per‑member SHA‑256 before extraction, rejects absolute / `..` tar members (traversal guard), and re‑tags the loaded image into a quarantined namespace so it can never overwrite `codeman/agent:base` or a pre‑existing tag.
|
||||
- **Host guard & the bridge‑hooks listener** — in‑container hook callbacks carry `Host: host.docker.internal` / `host.containers.internal`; both are on the always‑on host‑header allowlist (`DOCKER_HOST_GATEWAY_ALIASES`) and resolve to the host only from inside a container netns, so they are not a browser DNS‑rebinding surface. On a loopback‑only server, in‑container hooks are opt‑in via `CODEMAN_DOCKER_BRIDGE_HOOKS=1`, which binds a SECOND listener on the docker bridge gateway serving **only** the hook endpoints (every other path → `403`) into the same hook‑secret‑gated pipeline. The bridge is host‑internal (containers + host), not the LAN, so it does not widen network exposure; the hook secret is bind‑mounted read‑only and referenced by path.
|
||||
@@ -510,6 +516,7 @@ Full feature guide: [`docker-cases.md`](docker-cases.md).
|
||||
- **Auth is a parallel branch** (`middleware/auth.ts`) that leaves the single‑user path untouched: per‑user scrypt verify (`timingSafeEqual`, timing‑equalized against user enumeration), identity‑carrying cookies, a per‑username failure bucket (a botnet can't brute one account across IPs; one NATed user can't lock out the rest), and a `mustChangePassword` lockbox. The hook‑secret loopback bypass, host guard, and Origin/CSRF guard are unchanged (hooks authenticate the INSTANCE, not a user).
|
||||
- **Ownership is enforced server‑side only** and fails closed: `req.authUser` (a synthetic admin in single‑user), `findSessionOrFail` returns NOT_FOUND (never 403) for a foreign session, list/SSE/WS/file‑preview/search all filter by `session.owner`, and SSE routing defaults session‑scoped events to their owner (unresolved owner → withheld). The load‑bearing rule is **non‑admin `workingDir` confinement**: a non‑admin's session/one‑shot working dir must realpath‑resolve inside `~/codeman-users/<name>/cases`, checked BEFORE any disk write.
|
||||
- **Privileged actions are a one‑bit grant** (`canBypassPermissions`, default off): only granted users (and admins) get `--dangerously-skip-permissions` (others are silently downgraded to `--permission-mode auto`), shell‑mode sessions, cron `launchCommand`, and other CLIs' bypass flags. Machine‑level resources (remote/Docker host definitions, tunnel, self‑update, settings writes) are admin‑only.
|
||||
- **Clone Repo does not lend the server's git sign-in to non-admins.** A clone writes only inside the caller's own case space, so it is not admin-gated, but the server account's git credential helpers (the Docker image's opt-in `gh`/`az` helpers, or any `gh auth setup-git`) are shared by every user. A non-admin's clone and preflight therefore run with `git -c credential.helper=`, which empties the helper list including the URL-scoped entries (`cloneWithoutCredentialHelpers` in `case-routes.ts`, argv pinned in `test/git-clone.test.ts`). This closes the Clone Repo path only: the account's SSH keys still apply to an `ssh://` URL, and a non-admin's agent sessions run as the same account, consistent with the first bullet above. Docker cases are a second route to the same sign-in: with `CODEMAN_AGENT_IMAGE_INSTALL_GH`/`_AZ` on, a non-admin's Docker case with credential seeding on (the default) receives a copy of the server account's `gh`/`az` sign-in, exactly as it receives the Claude and Codex credentials.
|
||||
- **Admin actions are audited** append‑only to `~/.codeman/admin-audit.jsonl` (acting admin, action, target, IP). Passwords set by an admin create/reset are one‑time (returned once, force change). Under Basic auth, `logout` only truly ends QR‑issued sessions — to lock someone out, disable the account or reset the password (a proper login form is a deferred Phase 6).
|
||||
|
||||
---
|
||||
|
||||
@@ -0,0 +1,198 @@
|
||||
# Split-Pane Sessions — Design Spec
|
||||
|
||||
**Status**: Implemented (v1)
|
||||
**Author**: Claude (session with Tim), 2026-09-15
|
||||
**Scope**: v1 only. v2 items are named and explicitly deferred, not designed.
|
||||
|
||||
## Problem
|
||||
|
||||
Codeman's terminal area shows exactly one active session (pane) at a time —
|
||||
switching panes re-binds the single xterm instance and the single WebSocket
|
||||
to a different session. Multi-monitor spanning (`scripts/span-codeman.sh` /
|
||||
`span-codeman.ps1`) turned out to solve a different problem: it makes one
|
||||
browser window bigger, but that window still shows one session; floating
|
||||
subagent windows are draggable overlays on top of it, not tiled panes. There
|
||||
is no way today to see two live sessions (e.g. `w1-codeman` and
|
||||
`w1-mcp-memory`) side-by-side in one window, even on a monitor wide enough to
|
||||
fit both.
|
||||
|
||||
## Goal (v1)
|
||||
|
||||
From the active session, open a **second, independent, fully live session**
|
||||
in a pane beside it — draggable divider, side-by-side only. Closing the
|
||||
second pane collapses back to today's normal single-pane view. No
|
||||
persistence: a page reload always returns to single-pane. Floating
|
||||
subagent/Ultracode windows keep their current behavior unchanged (global,
|
||||
unconstrained across the whole viewport, split or not).
|
||||
|
||||
Explicitly out of scope for v1 (v2 candidates, not designed here):
|
||||
- More than 2 panes / grid layouts
|
||||
- Vertical (stacked) splits
|
||||
- Drag-a-tab-to-split as a trigger (v1 trigger is an explicit button + picker)
|
||||
- Persisting the split layout across reload or across devices
|
||||
- Mobile/tablet layouts (viewport is too narrow for this to make sense; gated
|
||||
to desktop widths the same way `home-sessions.js`'s rail is)
|
||||
- Feature parity between the two panes (see "Pane B is deliberately plainer"
|
||||
below)
|
||||
|
||||
## Current architecture (why this isn't a CSS change)
|
||||
|
||||
`terminal-ui.js` is built entirely around **singleton** state: `this.terminal`
|
||||
(one xterm instance), `this._ws`/`this._wsSessionId` (one WebSocket, rebound
|
||||
on every pane switch via `_disconnectWs()` + `_connectWs(newId)`), a
|
||||
`this._xtermSnapshots` map used only to restore scrollback into that one
|
||||
terminal when switching back to a session. Roughly 280 references to this
|
||||
singleton state exist across the file (input handling, resize/fit, sizing-
|
||||
token claims, mobile touch gestures, CJK IME, local-echo overlay wiring,
|
||||
keyboard accessory bar, link providers, etc.).
|
||||
|
||||
Showing two sessions at once therefore requires a second, independently
|
||||
alive xterm + WebSocket pair running concurrently — not a layout change to
|
||||
one shared instance.
|
||||
|
||||
**Related prior art**: `detachSession(id)` (app.js) already opens one session
|
||||
in a genuinely separate browser window (`isSoloWindow` mode) with its own
|
||||
independent WebSocket, and two of those can already be snapped side-by-side
|
||||
today with zero new code. That covers "two sessions visible at once" but not
|
||||
what this spec is for: one Codeman window with two panes and a divider you
|
||||
can drag without leaving your seat, each still a full participant in that
|
||||
window's floating subagent windows, header, and settings. This spec builds
|
||||
past detach, not a duplicate of it.
|
||||
|
||||
**Server-side check (done, not just assumed)**: `MAX_WS_PER_SESSION = 5`
|
||||
(`src/web/routes/ws-routes.ts`), scoped by `clientId:tabNonce`
|
||||
(`ws-connection-registry.ts`). Splitting always opens a *different* session
|
||||
in the second pane (self-splitting is disallowed, see below), so this is two
|
||||
sessions each getting their normal one connection — the existing cap is
|
||||
irrelevant here and needs no server change.
|
||||
|
||||
## Key design decision: Pane B is deliberately plainer than Pane A
|
||||
|
||||
Porting all ~280 singleton behaviors to a second, symmetric pane is not
|
||||
worth it for v1 — most of that code is input-quality-of-life for **mobile/
|
||||
touch** (local-echo overlay, CJK IME textarea, touch gesture handling,
|
||||
keyboard accessory bar), and this feature is desktop-only by nature (a split
|
||||
view needs a wide viewport). So:
|
||||
|
||||
- **Pane A** (the session that was already active when you opened the split)
|
||||
stays exactly what it is today — `this.terminal`, `this._ws`, unchanged
|
||||
code path, zero regression risk.
|
||||
- **Pane B** is a new, smaller `SplitTerminalPane` object: its own xterm
|
||||
instance + fit addon, its own WebSocket to `/ws/sessions/:id/terminal`,
|
||||
resize-on-divider-drag, and plain keyboard input. It does **not** get the
|
||||
local-echo overlay, CJK IME composition, touch/mobile handlers, or the
|
||||
keyboard accessory bar. On a desktop, typing directly into an xterm
|
||||
instance with no overlay is exactly how Codeman behaved before the local-
|
||||
echo overlay existed for touch devices — normal, not degraded, for a
|
||||
keyboard-and-mouse user.
|
||||
|
||||
If this asymmetry actually bothers you in daily use, promoting Pane B to full
|
||||
parity is a scoped v2 (extract the shared logic already once you have two
|
||||
call sites to compare, rather than guessing the right abstraction now).
|
||||
|
||||
One more asymmetry worth naming here rather than discovering by surprise:
|
||||
while both panes accept keyboard input, the global capture-phase shortcut
|
||||
handler (`app.js`) always resolves against Pane A — it has no notion of
|
||||
which pane currently has focus. So Ctrl+L or Ctrl+W typed while Pane B has
|
||||
focus clears or closes Pane A, not the session you were actually typing
|
||||
into. Not fixed for v1, same reasoning as the rest of this section.
|
||||
|
||||
## Components
|
||||
|
||||
### 1. `SplitTerminalPane` (new, `terminal-split.js`)
|
||||
|
||||
A small class, one instance per secondary pane:
|
||||
- `constructor(sessionId, mountEl)`
|
||||
- `connect()` — creates the xterm instance (same theme/font config as the
|
||||
primary, read from the same settings so it doesn't visually clash), opens
|
||||
`/ws/sessions/:id/terminal`, wires input → WS, WS → terminal write
|
||||
- `fit()` — calls the fit addon; called on divider drag (rAF-throttled) and
|
||||
on window resize
|
||||
- `destroy()` — disposes the xterm instance, closes the WS cleanly
|
||||
|
||||
No snapshot/scrollback-restore map is needed the way `_xtermSnapshots` exists
|
||||
for Pane A — Pane B is destroyed on close, not hidden-and-restored, since
|
||||
there's no persistence requirement.
|
||||
|
||||
### 2. Split container (layout)
|
||||
|
||||
```
|
||||
.terminal-split-container (flex row, only rendered when split is active)
|
||||
├── .terminal-wrap (existing element, Pane A — untouched)
|
||||
├── .split-divider (new, draggable seam)
|
||||
└── .terminal-pane-b (new, hosts SplitTerminalPane's xterm + a
|
||||
small header: session name + × close button)
|
||||
```
|
||||
|
||||
When not split, `.terminal-wrap` renders exactly as it does today (no
|
||||
wrapping container at all, to keep the no-split path byte-identical to
|
||||
current behavior). Splitting inserts the container and reparents
|
||||
`.terminal-wrap` into it as the first child — same reparenting pattern
|
||||
already used by `applySessionListLayout()` for `#sessionTabs`, so this isn't
|
||||
a new pattern for the codebase.
|
||||
|
||||
Default split is 50/50 (`flex-basis: 50%` each). Divider drag updates both
|
||||
panes' `flex-basis` live (rAF-throttled) and calls `fit()` on **both**
|
||||
terminals per tick, clamped to 20%/80% so neither pane can be dragged into an
|
||||
unusably thin sliver.
|
||||
|
||||
### 3. Trigger UI
|
||||
|
||||
A **"Split"** button (header, opt-in like the other header buttons —
|
||||
`showSplitButton`, default off, same pattern as `showMultiMonitorButton`)
|
||||
opens a small picker listing your other open sessions (reuses
|
||||
`this.sessions`/`sessionOrder`, filtered to exclude the currently active
|
||||
session — you cannot split a session against itself). Picking one:
|
||||
1. Creates the split container, reparents `.terminal-wrap`
|
||||
2. Instantiates `SplitTerminalPane` for the chosen session in `.terminal-pane-b`
|
||||
3. Button state flips to "close split" (or Pane B's own header × does it)
|
||||
|
||||
Closing (via Pane B's × or the header button toggling off):
|
||||
1. `SplitTerminalPane.destroy()`
|
||||
2. Removes `.terminal-split-container`, reparents `.terminal-wrap` back to
|
||||
its original location at 100% width
|
||||
3. Fires a resize/fit on Pane A (same `ResizeObserver`-driven fit already in
|
||||
place today — no new code needed here, it fires naturally once the
|
||||
container's size changes)
|
||||
|
||||
v2 note (not designed): dragging a session tab onto the active pane as an
|
||||
alternate trigger. You confirmed right-click doesn't work today (Codeman
|
||||
doesn't intercept it) and declined a keybind, so v1 is button+picker only.
|
||||
|
||||
### 4. Failure / edge cases
|
||||
|
||||
- **The Pane B session ends or is deleted while split is active** → treat
|
||||
identically to the user closing Pane B manually: destroy the pane, collapse
|
||||
to Pane A at full width.
|
||||
- **The Pane A session ends while split is active** → Pane B is promoted:
|
||||
it becomes the new single full-width pane (reusing today's normal
|
||||
single-pane code path means Pane B's `SplitTerminalPane` must hand off to
|
||||
a real `this.terminal`/`this._ws` binding — simplest correct approach is
|
||||
to just collapse the split and let normal session-select logic reopen
|
||||
Pane B's session as the new primary, rather than trying to promote the
|
||||
lightweight pane object in place).
|
||||
- **Both end** → falls through to today's normal "no active session" /
|
||||
welcome-screen state.
|
||||
- **Subagent/Ultracode floating windows** → no design work needed; they're
|
||||
already positioned independent of `.terminal-wrap`'s layout, so they
|
||||
continue to float over whichever pane(s) are on screen, unconstrained,
|
||||
exactly as today.
|
||||
|
||||
## Testing
|
||||
|
||||
- Unit: `SplitTerminalPane` connect/fit/destroy lifecycle (mock WS, like
|
||||
existing terminal tests use `TEST_PTY_SCRIPT`).
|
||||
- Route/integration: opening two WS connections to two different sessions
|
||||
from one simulated client concurrently — confirms the existing per-session
|
||||
cap and connection registry need no changes.
|
||||
- Browser (Playwright, `test/browser` since this is desktop-viewport-gated
|
||||
UI): open split via button+picker, verify both panes render live output
|
||||
independently, drag divider and confirm both refit, close Pane B and
|
||||
confirm Pane A returns to full width, kill the Pane B session externally
|
||||
and confirm auto-collapse.
|
||||
|
||||
## Open questions for review
|
||||
|
||||
None blocking — the scope-narrowing decisions above (Pane B feature parity,
|
||||
no persistence, side-by-side only, button+picker trigger) came directly from
|
||||
your answers during brainstorming. Flag anything here you want reconsidered.
|
||||
@@ -1,5 +1,10 @@
|
||||
# Tailscale Setup in the Installer (Plan)
|
||||
|
||||
> Superseded in part by [`installer-v2-plan.md`](installer-v2-plan.md) (2026-09-20), which
|
||||
> moved every human step before the build, added the sub-path / second-port answer for an
|
||||
> occupied `:443`, the opt-in rename, flags, and the done screen with a QR code. The
|
||||
> state machine and safety rules below still hold.
|
||||
|
||||
Goal: make "Codeman over Tailscale, with real HTTPS" a first-class, guided path in
|
||||
`install.sh`, instead of a one-line hint pointing at the docs. Today the safest
|
||||
recommended deployment (loopback bind + `tailscale serve`) is exactly what the
|
||||
|
||||
@@ -273,9 +273,16 @@ into the case's `.claude/settings.local.json` so that `/model` keeps working.
|
||||
- **Shell** for the times you want a terminal on your phone with no agent at all. It is a
|
||||
genuinely useful mode, not a fallback.
|
||||
|
||||
## Pointing one at your own server
|
||||
|
||||
Most of these harnesses can also run against a custom OpenAI-compatible endpoint instead of
|
||||
their native cloud backend, for one session at a time, an opt-in feature covered in full on
|
||||
[Custom Model Endpoints](Custom-Model-Endpoints).
|
||||
|
||||
## Read next
|
||||
|
||||
- [Core Concepts](Core-Concepts) - run modes versus location overlays.
|
||||
- [Custom Model Endpoints](Custom-Model-Endpoints) - run a harness against your own server.
|
||||
- [Settings Reference](Settings-Reference) - model, effort, and permission-mode settings.
|
||||
- [Keeping Agents Running](Keeping-Agents-Running) - what idle detection does per mode.
|
||||
- [Security](Security) - what skipping permission prompts actually means.
|
||||
|
||||
@@ -42,10 +42,17 @@ npm run typecheck
|
||||
npm run lint
|
||||
npm run format:check
|
||||
npm run check:frontend-syntax
|
||||
npm run check:browser-excludes
|
||||
npm test -- test/<file>.test.ts # one file, the normal way
|
||||
npm run test:ci # the full CI sweep
|
||||
```
|
||||
|
||||
`npm install` installs a `pre-push` git hook that runs the static checks above (about 10-40s,
|
||||
machine-dependent) and blocks a push that would fail them. It skips itself when you push
|
||||
something other than the checked-out HEAD, or when the tree has uncommitted changes the
|
||||
checks would read. Skip it once with `CODEMAN_SKIP_PREPUSH=1 git push`; a
|
||||
`pre-push` hook of your own is never overwritten.
|
||||
|
||||
**Never run bare `npm test`.** The default configuration includes browser-driven Playwright
|
||||
suites that need a live server, Chromium, and environment-specific baselines; they hang or
|
||||
fail on a normal machine. `test:ci` is the honest "run everything".
|
||||
|
||||
@@ -20,12 +20,19 @@ Three ways to get one, all under **+** next to the case picker:
|
||||
| How | Result |
|
||||
| ----------------- | ------------------------------------------------------------------------------------------------------ |
|
||||
| **Create New** | A fresh `~/codeman-cases/<name>` with a scaffolded `CLAUDE.md`. |
|
||||
| **Clone Repo** | A public repo cloned into `~/codeman-cases/<name>` and registered as a case. |
|
||||
| **Clone Repo** | A repo cloned into `~/codeman-cases/<name>` and registered as a case. Private repos need this machine's own git credentials (see below). |
|
||||
| **Link Existing** | An existing folder anywhere on disk, registered in place. Nothing is copied or moved. |
|
||||
|
||||
Linked cases keep living where they are. Deleting a case in Codeman removes the
|
||||
registration, and for a linked case that is all it removes.
|
||||
|
||||
**Clone Repo never asks for credentials.** It uses whatever the server's own git already has:
|
||||
an ssh key, or a credential helper such as `gh auth setup-git`. The Docker image can include
|
||||
helpers for GitHub (`gh`) and Azure DevOps (`az`), turned on in `docker-compose.override.yml`;
|
||||
then signing those CLIs in once from a shell session is enough. See the private repositories
|
||||
section of `docker/README.md`. Without credentials a private repo fails straight away with an
|
||||
authentication error.
|
||||
|
||||
**Cases created from scratch are the only copy of that code.** Uninstalling Codeman does not
|
||||
delete `~/codeman-cases/`, but treat that directory as real work, not scratch space.
|
||||
|
||||
|
||||
@@ -0,0 +1,189 @@
|
||||
# Custom Model Endpoints
|
||||
|
||||
Point a harness at your own OpenAI-compatible server instead of its native cloud backend, for
|
||||
one session at a time. "Custom endpoint" covers **local** hardware (llama.cpp, Ollama, vLLM,
|
||||
a home GPU rig, DGX Spark, Strix Halo) and **cloud** services (Azure AI Foundry's
|
||||
OpenAI-compatible endpoint, OpenRouter, a company gateway) alike, anything answering
|
||||
`GET /v1/models` and `POST /v1/chat/completions` in the standard shape.
|
||||
|
||||
**Off by default.** Turn it on in App Settings → Models → **Custom model endpoints**.
|
||||
|
||||
## Adding an endpoint
|
||||
|
||||
Still in App Settings → Models → Custom model endpoints:
|
||||
|
||||
1. **+ Add endpoint** — give it an id, a label, and the base URL (`http://192.168.1.50:8080`,
|
||||
say). An API key is optional; most local servers don't check one.
|
||||
2. **Discover** — fetches the endpoint's own model list over `GET /v1/models` and stores it.
|
||||
3. Pick a **default model** from what was discovered. This is the model the Run-menu entry
|
||||
applies directly when only one model is discovered; with two or more, it's just the one
|
||||
pre-marked in the picker dialog described below, not a silent default.
|
||||
|
||||
Endpoint management is admin-only in multi-user mode, the same as remote hosts and Docker
|
||||
hosts — these are machine-level infra, not a per-user setting.
|
||||
|
||||
**Model lists refresh themselves.** Every saved endpoint is re-discovered automatically every
|
||||
5 minutes in the background, so a model the server starts serving later — or stops serving —
|
||||
shows up without another manual click of **Discover**. One endpoint being unreachable on a
|
||||
given cycle (powered off, wrong network) never blocks the others from refreshing.
|
||||
|
||||
**Context length is picked up automatically where it can be, safely.** Against a
|
||||
llama.cpp/llama-swap server, discovery also learns each _currently loaded_ model's real
|
||||
context window and applies it to the launched session (Claude Code today — see below), so
|
||||
the harness stops assuming a large default window for a model name it doesn't recognise and
|
||||
overflowing a much smaller real one. It's deliberately never probed for a model that isn't
|
||||
already loaded, since asking a llama-swap server about an unloaded model can trigger an
|
||||
actual, slow model swap as a side effect — a model just not currently loaded keeps whatever
|
||||
context length an earlier cycle already learned for it instead.
|
||||
|
||||
## Running a session against one
|
||||
|
||||
With the setting on and at least one endpoint carrying a discovered model, the **Run**
|
||||
dropdown grows a **Custom Endpoints** section: one entry per harness that can redirect to a
|
||||
custom endpoint, per saved endpoint, e.g. "Claude Code (llama.cpp)". Picking one starts a
|
||||
session on that harness exactly the way its own entry would. It is a one-off "try this
|
||||
endpoint" action, not a sticky mode — the plain **Run** button still means "this harness,
|
||||
native cloud" afterward, and a fresh session never inherits whatever the last one was
|
||||
pointed at.
|
||||
|
||||
**Which model it uses depends on how many the endpoint has discovered.** With exactly one,
|
||||
the session launches straight away on that model — nothing to choose. With two or more, a
|
||||
small dialog asks which one to use for this launch before starting the session; the
|
||||
endpoint's default model, if set, is marked but not auto-picked, so a launch can deliberately
|
||||
use a different one without changing the saved default. The list is not raw discovery order
|
||||
either: the model llama-swap reports loaded and ready is moved to the top and tagged
|
||||
**Currently loaded**, and when nothing is loaded, the model you last launched on this harness
|
||||
and endpoint pair is moved up instead and tagged **Last used** (a per-device browser value, so
|
||||
another device starts from its own history). The default model keeps its own **Default** pill
|
||||
in both cases, and nothing is ever auto-chosen: the promoted row is simply the one under your
|
||||
thumb.
|
||||
|
||||
**For opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP, picking an entry launches
|
||||
straight onto the endpoint** — no restart, because the endpoint is applied before the
|
||||
session's process ever starts. **Claude still restarts the harness's process in place** —
|
||||
same tab, same conversation (`--resume`) — after a normal native launch, since that restart
|
||||
is far less jarring for Claude than for the other seven, whose own TUI can fully
|
||||
reinitialize on a restart. Either way, every supported harness reads its endpoint config at
|
||||
process start, never per turn, so there is no live hot-swap while a turn is running.
|
||||
|
||||
Picking an entry that launches a **brand-new** Claude session waits (up to 20 seconds) for it to
|
||||
finish its own startup before applying — a freshly started CLI reports itself as busy for its
|
||||
boot sequence, and applying to a genuinely busy session is refused so a real, in-progress
|
||||
turn is never interrupted out from under you. A session that is still busy after that wait
|
||||
(a very slow-starting CLI, or one you started typing into right away) surfaces that refusal
|
||||
as an ordinary error, which now stays on screen with a close button instead of vanishing
|
||||
after a few seconds — read it, it names the actual reason rather than a generic failure.
|
||||
|
||||
Entries are hidden entirely for a session in a **remote (SSH) or Docker case** — support for
|
||||
redirecting those hasn't landed yet, see below. The picker also only appears in the desktop
|
||||
**Run** dropdown; the phone home screen builds its own run picker separately and does not
|
||||
currently offer these entries.
|
||||
|
||||
**Against llama-swap, applying a selection also starts the actual model load, rather than
|
||||
waiting on your first prompt to do it.** llama-swap has no "switch model" button of its own
|
||||
— the only thing that starts a swap is a real request naming the model, and confirmed live:
|
||||
just applying a selection never reached llama-swap's own logs at all until something asked
|
||||
it to load. Picking an entry now also sends the smallest real request that will trigger
|
||||
that load, in the background, the moment the target model isn't already loaded and ready.
|
||||
|
||||
**The centred loading banner has no countdown and no automatic timeout — it waits as long as
|
||||
it takes, and tells you so.** When it knows the model's discovered file size (its GB figure,
|
||||
when llama-swap states one) it's shown too, e.g. "Loading qwen3.8-27b (16.4 GB) on
|
||||
llama-swap — this can take a while depending on your hardware and the model size." An
|
||||
earlier version tried to estimate and enforce a time limit, but real load time depends on
|
||||
hardware this feature has no way to know, so a fixed number was always a guess — worse, one
|
||||
that could kill a genuinely slow load partway through. If it really is taking too long, a
|
||||
**Cancel** button right on the banner ends the wait and **closes the session that load was
|
||||
for**, on your own call rather than a guessed deadline.
|
||||
|
||||
**The banner also shows a real, live second line of what llama.cpp itself is doing** — not
|
||||
a made-up progress phase, the actual next line the `llama-server` process printed, e.g.
|
||||
"llama.cpp: load_model: loading model '/models/.../Qwen3.8-27B.gguf'" then later
|
||||
"llama.cpp: llama_server: model loaded". It comes straight from llama-swap's own event
|
||||
feed, filtered down to just the backend process's own output (not llama-swap's own request
|
||||
logging), and stays on whatever it last said once the load goes quiet, rather than
|
||||
clearing back to nothing.
|
||||
|
||||
**You'll also be told if a session's model gets swapped out from under it later, not just
|
||||
at launch.** The conflict warning above only fires at the moment you launch or apply a
|
||||
model — llama.cpp only runs one model at a time, so if a DIFFERENT session using the same
|
||||
endpoint later triggers its own load, whatever was loaded before (including a session you
|
||||
already had running) gets silently evicted, with no warning at that instant since nothing
|
||||
conflicted when it was first set up. A background check (every 20 seconds) catches this
|
||||
after the fact and shows a toast naming which session lost its model and what's loaded now
|
||||
— so you know before typing into that session that it's about to reload (and, in turn,
|
||||
evict whatever displaced it).
|
||||
|
||||
**Claude Code specifically gets three extra fixes applied automatically:**
|
||||
|
||||
- Its discovered context length (see above) is passed through as
|
||||
`CLAUDE_CODE_MAX_CONTEXT_TOKENS`, so it doesn't send a full-size prompt against a much
|
||||
smaller real local context and overflow it.
|
||||
- Its session runs with an isolated `CLAUDE_CONFIG_DIR`, so the injected API key never sits
|
||||
in the same directory as a stored claude.ai login — that combination is harmless for actual
|
||||
requests (the API key wins) but the CLI still prints a "both claude.ai and
|
||||
ANTHROPIC_API_KEY set" warning about it, which this avoids entirely. The isolated directory
|
||||
keeps a link back to your real session history so the response viewer and similar features
|
||||
still work for that session. That isolated directory starts with no prior approvals of its
|
||||
own, so Codeman also pre-approves the injected key the same way answering Claude Code's own
|
||||
"Detected a custom API key" prompt once would — without it, that prompt would otherwise
|
||||
reappear on every single launch with nobody there to answer it.
|
||||
- **That same fresh isolated directory also looks like a brand-new Claude Code profile**, so
|
||||
without this fix it replayed the WHOLE first-run sequence every single launch: the theme
|
||||
picker, the security-notes screen, the "trust this folder?" dialog, and a one-time warning
|
||||
about running with permissions bypassed — none of which a real, already-used profile shows
|
||||
again. Codeman now pre-seeds that same "already been through this once" state (onboarding
|
||||
completed, this session's own project marked trusted, the bypass-permissions warning
|
||||
acknowledged) so a custom-model launch reaches the actual conversation exactly as fast as a
|
||||
native cloud one does, instead of stopping at a wizard with nobody there to click through it.
|
||||
|
||||
**If a model's real context is too small for Claude Code to even get started, you get a
|
||||
warning instead of a confusing failure.** Claude Code's own system prompt and tools take up
|
||||
roughly 40K tokens on their own, before you've typed anything — a small local model with a
|
||||
smaller real context than that fails outright on the very first message, no matter what
|
||||
context size Codeman tells it to expect (raising the declared context only changes when
|
||||
Claude Code trims _conversation history_, and there is none yet on message one). Picking
|
||||
such a model now shows an in-app dialog naming the model, its discovered context and what's
|
||||
needed, before anything launches or restarts, with the fix spelled out: reconfigure
|
||||
llama-swap to give that model (or a smaller one) an explicit larger context instead of
|
||||
relying on auto-fit (`--fit-ctx`), which sizes the context around fitting the biggest model
|
||||
rather than the biggest context — for example adding `-c 65536` to that model's llama-swap
|
||||
entry. "Launch anyway" is still there if you want to try regardless.
|
||||
|
||||
## Which harnesses actually work
|
||||
|
||||
| Harness | Status |
|
||||
| ---------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| **Claude Code, opencode, Pi, Grok, OMP** | Verified end-to-end against a real local server. |
|
||||
| **Codex** | Config is correct, and plain chat can work against a server that speaks the Responses API — but a real tool-call attempt comes back as inert text instead of running, so it's still not usable for real coding work. |
|
||||
| **Gemini** | Fails with an auth error gemini-cli raises once redirected. Unresolved; don't rely on it yet. |
|
||||
| **DeepSeek** | The original 404 is root-caused and fixed (DeepSeek Harness's own code was missing a `/v1` most local servers require) — not yet re-run against a real `dsh` install to confirm end-to-end. |
|
||||
| **Antigravity** | No known custom-endpoint mechanism at all. Not offered. |
|
||||
|
||||
Which harnesses show up in the Run-menu picker is read live off Codeman's own CLI registry,
|
||||
not a fixed list here, so this table can go stale before this page does — a greyed-out or
|
||||
missing entry is the more current answer.
|
||||
|
||||
## What it does not do
|
||||
|
||||
- **No remote or Docker sessions yet.** Both restart their agent differently under the hood
|
||||
(reattaching a durable tmux session rather than relaunching the process), so redirecting
|
||||
them needs its own plumbing that hasn't been built.
|
||||
- **No live hot-swap mid-conversation.** Applying a selection always restarts the process.
|
||||
- **No button to un-point a session from the UI yet.** Clearing back to native cloud is an
|
||||
HTTP call (`POST .../custom-model {"clear": true}`) or deleting the session; the settings
|
||||
panel manages saved endpoints, not what a running session is currently pointed at.
|
||||
- **Nothing is shared with your real cloud credentials.** The endpoint's own key, if any,
|
||||
never touches your Anthropic/OpenAI/Google login — a custom endpoint is a separate,
|
||||
explicit choice per session.
|
||||
|
||||
## Security
|
||||
|
||||
An endpoint's base URL can't point at a link-local or cloud-metadata address (both at save
|
||||
time and against the address it actually resolves to), the same guard Web Tabs uses for
|
||||
saved dashboards. Endpoint records and any per-session config files a harness needs are
|
||||
written with owner-only permissions. See
|
||||
[custom-model-endpoints-plan.md](https://github.com/Ark0N/Codeman/blob/master/docs/custom-model-endpoints-plan.md)
|
||||
in the repository for the full design reasoning, including why this feature closed a
|
||||
pre-existing gap in how session environment overrides were guarded rather than opening a new
|
||||
one.
|
||||
@@ -118,6 +118,41 @@ invisible from the host (`pi -c` and `grok -c` inside a docker case see only tha
|
||||
container's history). OMP's `sessions/` is the exception and is shared read-write, because
|
||||
Codeman reads it host-side for history and resume.
|
||||
|
||||
**Git hosts.** The agent image can also include the GitHub CLI (`gh`) and the Azure CLI (`az`,
|
||||
with the `azure-devops` extension), off by default, and its git then uses them as credential
|
||||
helpers for github.com and Azure DevOps. When the matching switch is on, their sign-ins are
|
||||
seeded like everything else, file by file: `~/.config/gh/hosts.yml` and `config.yml`, and the
|
||||
sign-in files from `~/.azure` (not its logs or extensions). With a switch off they are never
|
||||
copied in, even if the files exist. So once a switch is on and `gh auth login` / `az login`
|
||||
have been run where Codeman runs, agents in a Docker case can clone and push private repos on
|
||||
those hosts. Two limits:
|
||||
|
||||
- A token held in a desktop keyring or an encrypted token cache (Windows, macOS) is not
|
||||
inside those files and does not carry in. Sign in inside the container instead. The Docker
|
||||
server image and a headless Linux host keep it in the files, so they carry.
|
||||
- The sign-ins are mounted when a case container is **created**, so an existing container
|
||||
never picks them up. After turning a switch on, signing in, or rebuilding the agent image,
|
||||
**recreate the case container**: remove it, and the next session in that case creates a
|
||||
fresh one. (Or sign in inside the existing container instead.)
|
||||
|
||||
This hands a GitHub token and an Azure sign-in to every agent in a seeded Docker case, the
|
||||
same trust you already give it with Claude, Codex or gcloud. Turn seeding off for a case that
|
||||
should not have them.
|
||||
|
||||
Both CLIs are opt-in. To build the agent image with them, set
|
||||
`CODEMAN_AGENT_IMAGE_INSTALL_GH=1` and/or `CODEMAN_AGENT_IMAGE_INSTALL_AZ=1` where the image
|
||||
is built: in front of `node scripts/build-agent-image.mjs`, or in the Codeman server's
|
||||
environment for the image it builds automatically (in the Docker deployment, `environment:`
|
||||
in `docker-compose.override.yml`), then rebuild the image with `--no-cache`.
|
||||
`docker/README.md` ("Private repositories") has the details and the matching switches for
|
||||
the server image.
|
||||
|
||||
To give agents a fixed Git commit identity, set `CODEMAN_AGENT_IMAGE_GIT_USER_NAME` and
|
||||
`CODEMAN_AGENT_IMAGE_GIT_USER_EMAIL` together in that same environment (in the Docker
|
||||
deployment, set `GIT_USER_NAME` and `GIT_USER_EMAIL` in `docker/.env` instead, which feeds
|
||||
both images). An existing `codeman/agent:base` only picks it up after a `--no-cache` rebuild
|
||||
and recreated case containers; `docker/README.md` ("Git commit identity") has the details.
|
||||
|
||||
## Isolation
|
||||
|
||||
Every container runs hardened by default:
|
||||
|
||||
+47
-13
@@ -24,16 +24,22 @@ This installs Node.js, tmux and a build toolchain if they are missing (node-pty
|
||||
Linux prebuild, so it compiles from source), clones Codeman into `~/.codeman/app`, and
|
||||
builds it.
|
||||
|
||||
What it asks you:
|
||||
It starts by printing what it found (git, Node, tmux, build tools, agent CLIs, Tailscale,
|
||||
an existing install), then asks everything it needs up front, then does the work
|
||||
unattended. You can leave while it builds. What it asks you:
|
||||
|
||||
1. **Permission for every system change.** Package installs and agent CLI downloads are
|
||||
prompted individually. Nothing is installed silently. If no agent CLI is found, a menu
|
||||
offers to install any of them (DeepSeek excepted: its npm package installs only a
|
||||
launcher with no runnable profile), or you skip and install one yourself later.
|
||||
1. **One consent for the missing packages.** Git, Node.js, tmux and (on Linux) the build
|
||||
toolchain are installed after a single yes, and sudo asks for your password once for
|
||||
the whole run. Nothing is installed silently. If no agent CLI is found, a menu offers
|
||||
to install any of them (DeepSeek excepted: its npm package installs only a launcher
|
||||
with no runnable profile), or you skip and install one yourself later.
|
||||
2. **How the dashboard should be reachable.** Three choices:
|
||||
- **Tailscale** (recommended for phone access): keeps the loopback bind and walks you
|
||||
through `tailscale serve`, including the tailnet HTTPS toggle, then verifies the result
|
||||
end to end.
|
||||
- **Tailscale** (recommended for phone access): keeps the loopback bind, installs
|
||||
Tailscale if needed, logs in, enables the tailnet HTTPS toggle (it opens the admin
|
||||
page for you and waits; Ctrl+C there skips Tailscale for this run), then configures `tailscale serve` after the build and
|
||||
verifies the result end to end. If another app already owns `:443` on your node,
|
||||
you choose between a sub-path (`https://<machine>.<tailnet>.ts.net/codeman`, the
|
||||
default), a second port, replacing the other mapping, or skipping.
|
||||
- **Your local network** (`0.0.0.0`): prompts for a password. Skipping the password takes
|
||||
an explicit confirmation and ends on a loud warning.
|
||||
- **This machine only** (`127.0.0.1`): the safest option, and the default for a bare
|
||||
@@ -44,26 +50,54 @@ What it asks you:
|
||||
Tailscale. An existing loopback install defaults to keeping loopback, or to Tailscale when
|
||||
a serve mapping for Codeman is already there. A bare Enter never pulls in new software,
|
||||
and a non-interactive run always keeps the safe loopback default.
|
||||
3. **What to do when it finishes.** Run in this terminal, install as a background service
|
||||
that starts on boot, or do nothing yet.
|
||||
3. **What to call this machine on your tailnet** (Tailscale route only). By default the URL
|
||||
uses the machine's existing name. Answer yes to rename it `codeman-<hostname>`; the
|
||||
default is no, because the tailnet name is also what SSH and everything else on that
|
||||
machine are reached by.
|
||||
4. **Whether to run Codeman in the background.** Enter installs a systemd user service or a
|
||||
macOS LaunchAgent that starts on boot; answering no offers to start it in this terminal
|
||||
instead, or not at all.
|
||||
|
||||
It ends on a screen with the URL (your tailnet, your network, or this machine), a QR code to
|
||||
scan with your phone, and the two commands you need to manage the service.
|
||||
|
||||
Re-running the same one-liner **updates an existing install in place**. Local changes in
|
||||
`~/.codeman/app` are stashed rather than discarded, a running service is restarted and
|
||||
verified, and your existing network binding is preserved. An interrupted first install
|
||||
resumes instead of restarting.
|
||||
|
||||
Two other entry points exist:
|
||||
Other entry points:
|
||||
|
||||
```bash
|
||||
install.sh status # print the URLs, the QR code and the manage commands again
|
||||
install.sh update # update only
|
||||
install.sh uninstall # remove
|
||||
install.sh uninstall # remove (offers to undo a rename it performed)
|
||||
install.sh tailscale # retrofit Tailscale access onto an existing install
|
||||
install.sh name [<n>] # rename this machine on your tailnet (default codeman-<hostname>)
|
||||
install.sh cloudflared # install cloudflared for the in-app Cloudflare tunnel
|
||||
```
|
||||
|
||||
**Flags** answer the questions from the command line and pipe through `bash -s --`:
|
||||
|
||||
```bash
|
||||
curl -fsSL https://getcodeman.com/install | bash -s -- --tailscale --service
|
||||
curl -fsSL https://getcodeman.com/install | bash -s -- --lan --password 'x' --service
|
||||
curl -fsSL https://getcodeman.com/install | bash -s -- --local --run
|
||||
```
|
||||
|
||||
`--tailscale` / `--lan` / `--local` answer the access question, `--name <n>` / `--no-rename`
|
||||
the name, `--service` / `--run` / `--no-start` the last one. `--yes` takes every default
|
||||
(it still waits on a Tailscale login URL, and a network bind still asks for a password).
|
||||
`--port <n>` moves Codeman off 3000; the service file and the serve mapping follow it. On an
|
||||
existing install, `--port` and `--password` re-run the setup so the service file picks them up,
|
||||
and a re-run with `--lan` or `--tailscale` keeps the password the service already has.
|
||||
|
||||
**Automation and CI**: with no terminal attached, any step that would change the system
|
||||
aborts with instructions instead of running silently. Set `CODEMAN_NONINTERACTIVE=1` to
|
||||
approve those steps. `CODEMAN_TAILSCALE=1` preselects the Tailscale answer, and never
|
||||
installs Tailscale itself non-interactively.
|
||||
installs Tailscale itself non-interactively; a non-interactive run never renames the
|
||||
machine and never starts a service. Everything the unattended steps print goes to
|
||||
`~/.codeman/install.log`, and the last lines of it are shown when a step fails.
|
||||
|
||||
## Route B: npm
|
||||
|
||||
|
||||
@@ -34,6 +34,8 @@ Press `Ctrl+?` in the app for the same list in a floating overlay.
|
||||
| Right-click | Copy the selection. With nothing selected the native menu is left alone. |
|
||||
| `Ctrl+Z` | Swallowed in agent sessions so a running CLI cannot be suspended. Normal job control in a shell. |
|
||||
|
||||
Anything you copy is cleaned on the way to the clipboard: each line loses the padding spaces a full-screen program paints across the rest of the row. Leading indentation is left exactly as it is, so indented code, a `git log` message body and `git diff` context lines paste back the way they looked on screen. An `Alt+drag` rectangular selection is copied exactly as it looks, so its columns stay lined up.
|
||||
|
||||
## Everything else
|
||||
|
||||
| Shortcut | Action |
|
||||
|
||||
@@ -60,14 +60,20 @@ On by default; it can be turned off in settings.
|
||||
|
||||
A row of keys above the virtual keyboard, and what it contains depends on the session.
|
||||
|
||||
**Agent sessions** get quick actions: `/init`, `/clear`, `/compact`, a clipboard key, `Esc`,
|
||||
a path picker, an image key, and 🧠 when Read My Mind is on. Destructive commands need a
|
||||
double press, so you cannot fire `/clear` with a stray thumb. On Codex sessions the bar also
|
||||
shows `⇧←` and `⇧→`, the Shift-modified arrows Codex binds to editing the last queued
|
||||
message and walking the prompt stack.
|
||||
**Agent sessions** get quick actions: `/init`, `/clear`, `/compact`, a Compose key, `Esc`,
|
||||
a path picker, and 🧠 when Read My Mind is on. Compose opens a multiline editor with
|
||||
autocorrect: Enter adds a new line, and only Send delivers the text, as one paste followed
|
||||
by Enter, so your line breaks reach the agent intact. Anything already typed on the terminal
|
||||
prompt moves into the editor when it opens. Drafts are kept per session and in memory only,
|
||||
so switching tabs keeps them and a page reload forgets them; a dot on the key shows a draft
|
||||
is parked. The editor's Image button attaches photos and puts their paths into the draft.
|
||||
Destructive commands need a double press, so you cannot fire `/clear` with a stray thumb. On
|
||||
Codex sessions the bar also shows `⇧←` and `⇧→`, the Shift-modified arrows Codex binds to
|
||||
editing the last queued message and walking the prompt stack.
|
||||
|
||||
**Shell sessions** automatically swap it for terminal controls: `Ctrl`, `Esc`, `Tab`, four
|
||||
arrows, paste, and dismiss. Your normal preference is remembered and restored when you
|
||||
arrows, a direct Paste key (shell input is not an agent prompt, so there is no Compose
|
||||
there), and dismiss. Your normal preference is remembered and restored when you
|
||||
switch back to an agent session, so a settings change during a shell session cannot strip
|
||||
the bar away permanently.
|
||||
|
||||
|
||||
@@ -103,6 +103,34 @@ locked phone and the agent continues.
|
||||
With the inbox off, the buttons are stripped from the notification payload entirely rather
|
||||
than being shown and failing.
|
||||
|
||||
## When a session is watching its own work
|
||||
|
||||
An agent that starts a monitor, puts a shell in the background or hands a task to a cloud
|
||||
session is told by its CLI to end the turn and wait to be notified. The pane then goes
|
||||
quiet, and the CLI's idle notification arrives about a minute later — for a session that
|
||||
wants nothing from you.
|
||||
|
||||
Codeman reads what the CLI prints about its own background work and treats that prompt
|
||||
differently. It raises no tab alert, no desktop notification and no push, the session stays
|
||||
out of NEEDS YOU on every surface, and the row wears a blue **watching** badge instead. Hover
|
||||
it, or read it on a phone through your screen reader, and it says what is running: "1
|
||||
monitor", "2 shells", "1 background terminal".
|
||||
|
||||
The prompt itself is not thrown away. It sits in the Approvals drawer as an ordinary card,
|
||||
still answerable, with a line reading "quiet, watching 1 monitor" where a card you had
|
||||
already looked at would say nothing. The next time that session goes quiet for an ordinary
|
||||
reason, it alerts you exactly as before.
|
||||
|
||||
Two limits are worth knowing. A permission prompt or a question dialog still goes red
|
||||
whatever else the agent started, because that one blocks it outright. A question asked in
|
||||
plain prose is not a dialog, so an agent that starts a monitor and then writes "which branch
|
||||
should I target?" is quiet along with the rest — check a watching session yourself if it has
|
||||
been quiet longer than the work it is waiting for should take.
|
||||
|
||||
An agent waiting for your comments on an artifact it published never counts as watching.
|
||||
Claude shows that as "1 Artifact comment monitor", but the agent hears nothing until you
|
||||
comment, so the session alerts you like any other quiet session.
|
||||
|
||||
## The phone overview
|
||||
|
||||
On phones, tapping the "C" logo gives a session overview with **NEEDS YOU** first, then
|
||||
|
||||
@@ -43,7 +43,7 @@ To make a new one, click **+** next to the picker. The Add Case dialog has three
|
||||
| Tab | Use it when |
|
||||
| ----------------- | ------------------------------------------------------------------------------------------------------------------ |
|
||||
| **Create New** | Starting a fresh project. Creates `~/codeman-cases/<name>` and scaffolds a `CLAUDE.md` into it. |
|
||||
| **Clone Repo** | Working on an existing public repo. Paste the URL; Codeman preflights it as you type, offers the repo's real branches and tags, and fills in the case name. |
|
||||
| **Clone Repo** | Working on an existing repo: public, or private once this machine's git can authenticate (the Docker image can include `gh`/`az` helpers for this). Paste the URL; Codeman preflights it as you type, offers the repo's real branches and tags, and fills in the case name. |
|
||||
| **Link Existing** | The code is already on disk. Point at the folder, with **Browse** if you would rather click than type. |
|
||||
|
||||
The gear next to the picker holds two per-case toggles: **Agent Teams** and
|
||||
|
||||
@@ -33,11 +33,12 @@ Your devices join a private network, and Codeman stays bound to loopback. Nothin
|
||||
published to the internet, and you get real HTTPS with a real certificate.
|
||||
|
||||
The installer sets this up for you, including installing Tailscale, logging in, enabling
|
||||
tailnet HTTPS, and verifying the result end to end. To retrofit it onto an existing
|
||||
install:
|
||||
tailnet HTTPS, and verifying the result end to end. It ends on the URL with a QR code to
|
||||
scan. To retrofit it onto an existing install, or to see the URL and QR code again:
|
||||
|
||||
```bash
|
||||
install.sh tailscale
|
||||
install.sh status
|
||||
```
|
||||
|
||||
By hand:
|
||||
@@ -49,6 +50,22 @@ tailscale serve status
|
||||
|
||||
Then open `https://<machine>.<tailnet>.ts.net` from any device on your tailnet.
|
||||
|
||||
### The name in the URL
|
||||
|
||||
The URL is the machine's MagicDNS name, so on a machine called `tnode` it is
|
||||
`https://tnode.<tailnet>.ts.net`. Three ways to influence that, from least to most work:
|
||||
|
||||
| You want | How |
|
||||
| ------------------------------------------ | ----------------------------------------------------------------------------------------------------- |
|
||||
| The machine's existing name (default) | Nothing. This is what the installer does unless you say otherwise. |
|
||||
| `https://codeman-<hostname>.<tailnet>.ts.net` | Answer yes to the installer's name question, pass `--name codeman-<hostname>`, or run `install.sh name`. This renames the machine tailnet-wide (SSH included), which is why the installer defaults to no. `install.sh uninstall` offers to rename it back. |
|
||||
| `https://codeman.<tailnet>.ts.net` | A [Tailscale Service](https://tailscale.com/docs/features/tailscale-services). Only a **tagged** node can host one (a device signed in with a user account cannot), the service is defined and approved in the admin console, and the feature is in beta. The installer does not set this up; it is a `tailscale serve --service=svc:codeman --https=443 127.0.0.1:3000` on a tagged host once the service exists. |
|
||||
|
||||
If `:443` on your node already belongs to another app, the installer offers Codeman under
|
||||
`https://<machine>.<tailnet>.ts.net/codeman` (the default, via `tailscale serve --set-path`
|
||||
plus Codeman's `--base-url`), on a second port (`https://<machine>.<tailnet>.ts.net:8443`),
|
||||
or replacing the other mapping. It never replaces anything without asking.
|
||||
|
||||
Notes:
|
||||
|
||||
- Keep the loopback bind. `tailscale serve` connects to `127.0.0.1:3000` locally, so
|
||||
@@ -58,7 +75,11 @@ Notes:
|
||||
- Codeman's Host-header allowlist already accepts `.ts.net`, so no extra configuration is
|
||||
needed.
|
||||
- The installer never resets or rewrites `serve` mappings other than the one pointing at
|
||||
Codeman's port, so unrelated serve configuration is left alone.
|
||||
Codeman's port, so unrelated serve configuration is left alone. It also never opens a
|
||||
`tailscale funnel` (that is the public internet) and never advertises a Tailscale Service.
|
||||
- On macOS, the App Store and standalone Tailscale apps only run once someone is logged in,
|
||||
so a headless Mac needs the open-source `tailscaled` for the URL to come back after a
|
||||
reboot on its own.
|
||||
|
||||
## Cloudflare tunnel
|
||||
|
||||
|
||||
@@ -67,6 +67,14 @@ On Linux, if you want the service running while you are not logged in:
|
||||
loginctl enable-linger $USER
|
||||
```
|
||||
|
||||
On macOS, a LaunchAgent starts when you log in, not at boot. A headless Mac (no GUI login)
|
||||
needs a system LaunchDaemon instead, written by hand as root. The installer recognises an
|
||||
existing `/Library/LaunchDaemons/com.codeman.web.plist` and leaves it alone rather than
|
||||
installing a LaunchAgent next to it, since the two would fight over the port; remove the
|
||||
daemon first if you want to switch. The same login caveat applies to the App Store and
|
||||
standalone Tailscale apps, so on a headless Mac the Tailscale URL only comes back after a
|
||||
reboot if the open-source `tailscaled` is used.
|
||||
|
||||
### Writing the unit by hand
|
||||
|
||||
**Linux (systemd user unit):**
|
||||
|
||||
@@ -46,6 +46,7 @@ supervised by systemd or launchd; npm installs report as non-updatable. See
|
||||
| Extended Keyboard Bar | Per device | Which accessory bar phones get. Shell sessions override it while they are active. |
|
||||
| Wheel Scrolls Local History | Off | Keeps the wheel on the local buffer instead of forwarding it to the CLI. |
|
||||
| Auto Copy Selection | Off | Copies highlighted terminal text to the clipboard the moment you finish selecting it. Ctrl+C still copies on demand. |
|
||||
| Trim The Pane Margin On Copy | On | Takes the left margin a full-screen agent CLI paints down its own edge off a copy, so the text pastes flush. Each CLI declares its own width, and the strip never exceeds the indent every selected line shares, so nesting is kept. Claude Code and Codex declare a margin; a shell does not. |
|
||||
| Normal / Bold font weight | xterm defaults | Per device, each slot from 100 to 900. The bundled JetBrains Mono renders every step, so a lighter normal weight makes Claude's bold headings stand out. Applies live to the terminal, both echo overlays and open team panes. |
|
||||
| WebGL Renderer | On | With a GPU-stall watchdog that falls back to DOM rendering. |
|
||||
| Gesture Control | Off | Camera hand tracking. Also needs `CODEMAN_GESTURE=1` on the server. |
|
||||
@@ -55,12 +56,14 @@ supervised by systemd or launchd; npm installs report as non-updatable. See
|
||||
Chips for every optional header control, with a live preview of the resulting header:
|
||||
|
||||
Run, Font Size, System Stats, Redraw Terminal, Response Viewer, Away Digest, Session
|
||||
Manager, Attachments, File Viewer, Multi-monitor, Plan Usage, Lifecycle Log, Monitor,
|
||||
Manager, Attachments, File Viewer, Multi-monitor, Split, Plan Usage, Lifecycle Log, Monitor,
|
||||
Project Insights, File Browser, Subagents, Approvals Inbox, Read My Mind, Ultracode Agents,
|
||||
Ultracode Windows, Cron.
|
||||
|
||||
Most default to off. The stock desktop header is system stats, File Viewer, and the gear.
|
||||
New header controls never appear on phones.
|
||||
New header controls never appear on phones. Split is desktop-only regardless of this
|
||||
setting — the button and the feature both stay off below a ~1180px viewport, where two
|
||||
resizable panes plus their divider have nowhere to go.
|
||||
|
||||
This section also holds background-agent tracking, including whether to track agents for
|
||||
every session or only the active tab.
|
||||
@@ -92,6 +95,10 @@ Model and effort are both **soft defaults**: the model is written into the case'
|
||||
`.claude/settings.local.json` and effort is passed at start, so `/model` and `/effort`
|
||||
inside a session override them at any time.
|
||||
|
||||
**Custom model endpoints** (off by default) adds a saved-endpoint list plus a matching
|
||||
section to the Run dropdown, for pointing a harness at your own OpenAI-compatible server
|
||||
instead of its native cloud backend. See [Custom Model Endpoints](Custom-Model-Endpoints).
|
||||
|
||||
### Agents & CLIs
|
||||
|
||||
| Setting | Notes |
|
||||
|
||||
@@ -48,6 +48,7 @@ One tab per session, in your order, and that order syncs across your devices.
|
||||
| Yellow tab, blinking | The agent is waiting for input from you. |
|
||||
| Red tab, blinking | A question or permission prompt is blocking the session. |
|
||||
| No dot | The session is not running. |
|
||||
| Muted grey dot plus an `exited (137)` badge | The agent inside the pane has exited, with that exit code (or `exited (signal 9)`). A bare `exited` means tmux saw the pane die but did not report how, which is not the same as a clean `exited (0)`. Detailed sidebar and rail rows read `exited` in their pill. |
|
||||
|
||||

|
||||
|
||||
@@ -116,6 +117,7 @@ The right side of the header. Almost all of these are off until you enable them
|
||||
| Lifecycle Log | Off | Session start, exit, and kill audit trail. |
|
||||
| Cron ⏰ | Off | Scheduled jobs. |
|
||||
| Multi-monitor | Off, macOS | Opens a window spanning every display. |
|
||||
| Split | Off, desktop only | View a second session beside the active one, with a draggable divider. |
|
||||
| Tunnel indicator | When a tunnel runs | Cloudflare tunnel status. |
|
||||
| Admin panel | Multi-user only | User administration. |
|
||||
|
||||
@@ -149,10 +151,13 @@ Worth knowing:
|
||||
|
||||
- **Scrollback.** Agent/TUI sessions pull their entire tmux scrollback on first open.
|
||||
Shell sessions open from a bounded recent tail so a large transcript cannot stall tab
|
||||
switching; press **Load full history** to pull the rest explicitly. Ordinary Shell scrolling
|
||||
and automatic output recovery stay within the bounded browser buffer.
|
||||
- **Wheel and touch scrolling** are forwarded into Claude's own transcript on recent Claude
|
||||
versions, so the wheel scrolls the conversation rather than the terminal. `Shift+Wheel` is
|
||||
switching. Scrolling to the top of a Shell pane pulls the most recent 1 MiB of its tmux
|
||||
history; press **Load full history** to pull the rest explicitly. Automatic output
|
||||
recovery stays within the bounded browser buffer.
|
||||
- **Wheel and touch scrolling** are forwarded into Claude's own transcript when a recent
|
||||
Claude runs fullscreen (`CLAUDE_CODE_NO_FLICKER=1`, or `"tui": "fullscreen"` in
|
||||
`~/.claude/settings.json`), so the wheel scrolls the conversation rather than the terminal.
|
||||
Claude's default inline view keeps its history in the terminal and scrolls locally. `Shift+Wheel` is
|
||||
always local scrollback. Other CLIs scroll locally.
|
||||
- **Selection copy.** `Ctrl+C` copies when text is selected and interrupts when it is not.
|
||||
`Ctrl+Shift+C` always copies.
|
||||
|
||||
@@ -164,8 +164,11 @@ Scrollback behaviour depends on the CLI, and Codeman adjusts what it strips per
|
||||
Things to try:
|
||||
|
||||
- `Shift+Wheel` always scrolls the local buffer, whatever else is going on.
|
||||
- On Claude sessions with a recent CLI, the wheel is forwarded into Claude's own transcript,
|
||||
so it scrolls the conversation rather than the terminal buffer. That is intended.
|
||||
- On Claude sessions running fullscreen (recent CLI with mouse tracking on), the wheel is
|
||||
forwarded into Claude's own transcript, so it scrolls the conversation rather than the
|
||||
terminal buffer. That is intended. Claude's default inline view scrolls locally; turn
|
||||
fullscreen on with `CLAUDE_CODE_NO_FLICKER=1` or `"tui": "fullscreen"` in
|
||||
`~/.claude/settings.json`.
|
||||
- Scrolling to the very top pulls the full tmux scrollback again on demand.
|
||||
|
||||
### The wheel does nothing in a Codex session
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
|
||||
- [The Dashboard](The-Dashboard)
|
||||
- [Agent CLIs](Agent-CLIs)
|
||||
- [Custom Model Endpoints](Custom-Model-Endpoints)
|
||||
- [Working With Files](Working-With-Files)
|
||||
- [Input And Voice](Input-And-Voice)
|
||||
- [Mobile Guide](Mobile-Guide)
|
||||
|
||||
+1479
-554
File diff suppressed because it is too large
Load Diff
Generated
+3
-2
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "aicodeman",
|
||||
"version": "1.30.0",
|
||||
"version": "1.33.2",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "aicodeman",
|
||||
"version": "1.30.0",
|
||||
"version": "1.33.2",
|
||||
"hasInstallScript": true,
|
||||
"license": "MIT",
|
||||
"workspaces": [
|
||||
@@ -55,6 +55,7 @@
|
||||
"@types/web-push": "^3.6.4",
|
||||
"@types/ws": "^8.18.1",
|
||||
"@vitest/coverage-v8": "^4.1.8",
|
||||
"@xterm/headless": "^6.0.0",
|
||||
"agent-browser": "^0.6.0",
|
||||
"esbuild": "^0.27.3",
|
||||
"eslint": "^9.0.0",
|
||||
|
||||
+3
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "aicodeman",
|
||||
"version": "1.30.0",
|
||||
"version": "1.33.2",
|
||||
"description": "Mission control for AI coding agents - run 20 autonomous agents with real-time monitoring and session persistence",
|
||||
"type": "module",
|
||||
"main": "dist/index.js",
|
||||
@@ -28,6 +28,7 @@
|
||||
"pretest:mobile": "node scripts/prepare-test-vendor.mjs",
|
||||
"test:mobile": "vitest run --config test/mobile/vitest.config.ts",
|
||||
"check:frontend-syntax": "node scripts/check-frontend-syntax.mjs",
|
||||
"check:browser-excludes": "node scripts/check-browser-test-excludes.mjs",
|
||||
"fix:node-pty": "node scripts/fix-node-pty.mjs",
|
||||
"typecheck": "tsc --noEmit && tsc -p config/tsconfig.scripts.json",
|
||||
"lint": "eslint --config config/eslint.config.js 'src/**/*.ts'",
|
||||
@@ -123,6 +124,7 @@
|
||||
"@types/web-push": "^3.6.4",
|
||||
"@types/ws": "^8.18.1",
|
||||
"@vitest/coverage-v8": "^4.1.8",
|
||||
"@xterm/headless": "^6.0.0",
|
||||
"agent-browser": "^0.6.0",
|
||||
"esbuild": "^0.27.3",
|
||||
"eslint": "^9.0.0",
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "codeman",
|
||||
"description": "Drive Codeman, the self-hosted session manager for AI coding agents, from inside a Claude Code session: spawn worker sessions, prompt them, wait for them, read their answers, clean up. Acts only inside a Codeman-managed session.",
|
||||
"version": "1.30.0",
|
||||
"version": "1.33.2",
|
||||
"author": {
|
||||
"name": "Ark0N",
|
||||
"url": "https://github.com/Ark0N"
|
||||
|
||||
@@ -47,7 +47,7 @@ later call opens with, and your first REAL call performs them anyway:
|
||||
|
||||
```bash
|
||||
. "${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh" 2>/dev/null
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
```
|
||||
|
||||
⚠️ **Never spend a Bash call on this check alone.** §1's block opens with this same
|
||||
@@ -75,8 +75,8 @@ PRE="${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh"
|
||||
mkdir -p "$(dirname "$PRE")"
|
||||
# Rewrite unless the file already ends with THIS version's stamp, so a stale or a
|
||||
# half-written file self-heals here instead of costing you a round trip to rm it.
|
||||
grep -qs '^CODEMAN_PREAMBLE=1.22.0$' "$PRE" || (umask 077; cat > "$PRE" <<'PREAMBLE'
|
||||
# ---- Codeman agent preamble 1.22.0 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
grep -qs '^CODEMAN_PREAMBLE=1.30.1$' "$PRE" || (umask 077; cat > "$PRE" <<'PREAMBLE'
|
||||
# ---- Codeman agent preamble 1.30.1 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
API="${CODEMAN_API_URL:?CODEMAN_API_URL not set; refusing to guess}"
|
||||
SELF="${CODEMAN_SESSION_ID:?CODEMAN_SESSION_ID not set}"
|
||||
# Credentials, cheapest first. Your session has usually INHERITED the server's
|
||||
@@ -150,6 +150,27 @@ _trust_key() { # <sid> -> "confirm" | "move" | "" (nothing safe to press)
|
||||
| tr -d ' \t' | grep -i '❯[0-9.]*\(yes,itrustthisfolder\|no,exit\)' | tail -1 \
|
||||
| sed -e 's/.*[Yy]es,.*/confirm/' -e 's/.*[Nn]o,.*/move/'
|
||||
}
|
||||
# ---- the composer: is the prompt still sitting there, unsent? ----
|
||||
# ⚠️ Claude Code 2.1.277 (auto-installed 2026-09-18) takes typed text the moment the
|
||||
# composer paints but IGNORES Enter for the first 30-50 seconds after it: the \r that
|
||||
# Codeman sends 50 ms after the text and a lone nudge at 20 s both leave the prompt
|
||||
# stranded, with `0 tokens`, while the wait burns its whole timeout. Measured through
|
||||
# this very route: Enter at 28 s stranded, Enter at 51 s submitted. So sendwait READS
|
||||
# the composer and keeps pressing Enter while the prompt is still there.
|
||||
_composer_text() { # <sid> -> the composer's text with ALL whitespace removed: "" once
|
||||
# the prompt was taken, "?" when the pane shows no composer at all. The composer is
|
||||
# the LAST `❯` line: Claude Code echoes a submitted prompt with the same glyph higher
|
||||
# up in the transcript, so only the last one says whether the text was taken.
|
||||
local t
|
||||
t=$("${CURL[@]}" -G "$API/api/v1/sessions/$1/terminal" --data-urlencode 'full=1' \
|
||||
| jq -r '.data.terminalBuffer // empty' \
|
||||
| sed -e "s/$(printf '\033')\[[0-9;?]*[a-zA-Z]//g" -e "s/$(printf '\033')[()][AB0]//g" \
|
||||
| tr -d '\r' | grep -a '^[[:space:]]*❯' | tail -1)
|
||||
[ -n "$t" ] || { printf '?'; return 0; }
|
||||
# Claude Code draws a NO-BREAK SPACE (U+00A0) after the glyph, which [:space:] does
|
||||
# not cover, so it is stripped by its bytes, portably (BSD sed has no \xHH).
|
||||
printf '%s' "$t" | sed 's/^[[:space:]]*❯//' | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g"
|
||||
}
|
||||
_accept_trust() { # <sid> -> 0 once it has answered the dialog, 1 if it could not
|
||||
local sid="$1" k i=1
|
||||
while [ "$i" -le 6 ]; do
|
||||
@@ -262,12 +283,16 @@ spawn_workers() {
|
||||
# worker a silent no-op that still "succeeds" and reports the previous turn's state.
|
||||
# Pass seq explicitly for exactly one reason: resending a possibly-delivered frame as a
|
||||
# deliberate duplicate, at the SAME number (§5.3).
|
||||
# Delivery is SELF-HEALING: an Ink repaint occasionally eats the Enter, leaving the
|
||||
# typed prompt stranded on the composer while a long wait runs its whole timeout
|
||||
# (observed live). So the first wait is short; on its timeout a bare \r goes out (the
|
||||
# missing Enter when the prompt is stranded, a no-op when the turn is genuinely
|
||||
# running), then the ORIGINAL frame is resent unchanged, which the server takes as a
|
||||
# tagged duplicate: it re-waits without retyping (§5.3). Trustworthy for a worker
|
||||
# Delivery is SELF-HEALING: the Enter can be lost (an Ink repaint eats it, and Claude
|
||||
# Code 2.1.277+ ignores it outright for the first 30-50 s after the composer paints),
|
||||
# leaving the typed prompt stranded on the composer while a long wait runs its whole
|
||||
# timeout (observed live, twelve reviews in a row). So the first wait is short; on its
|
||||
# timeout the ORIGINAL frame is resent unchanged as a long re-wait (a tagged duplicate:
|
||||
# the server re-waits without retyping, §5.3) and kept open in the background, while
|
||||
# the composer is READ (_composer_text) and, as long as the prompt is still sitting
|
||||
# there, a bare \r goes out about every ten seconds, up to twelve times. An empty
|
||||
# composer ends the loop, so a prompt that was taken is never nudged again, and the
|
||||
# wait that was open the whole time is what reports the turn's end. Trustworthy for a worker
|
||||
# spawn_worker handed back -- claude (hooks vetted) or deepseek (status bridge) --
|
||||
# and for those only. Hook-less workspaces and the other modes resolve on flapping
|
||||
# idle: markers instead (§5.5). ⚠️ A dsh worker running a profile that does not
|
||||
@@ -275,7 +300,7 @@ spawn_workers() {
|
||||
# it accepts the send and then burns both waits. One timeout on a dsh worker whose
|
||||
# pane clearly finished means that profile, so switch that worker to markers.
|
||||
sendwait() {
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r c head n=0 tmp bg i
|
||||
# `wait:"stop,exit"`, never the `wait:true` default set: that set also carries
|
||||
# `idle`, which is INFERRED from output stabilization and flaps mid-turn. On a
|
||||
# dsh worker whose TUI repaints rarely the session reads `idle` while the model
|
||||
@@ -290,16 +315,38 @@ sendwait() {
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$body")
|
||||
if jq -e '.data.delivered and .data.wait.timedOut' <<<"$r" >/dev/null 2>&1; then
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
# The resend is a tagged DUPLICATE, so the server skips the write and reports
|
||||
# `delivered:false` for it -- truthfully, but about the wrong send. The first
|
||||
# one delivered, so carry that forward, or §1's cleanup reads a completed turn
|
||||
# as an undelivered one and keeps a finished worker forever.
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" \
|
||||
| jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end')
|
||||
# ⚠️ The long re-wait is registered FIRST and stays open for the rest of this call,
|
||||
# in the background, while the Enter loop below works the composer. Signals have
|
||||
# no history: a `stop` that fires while no wait is open (during a composer read
|
||||
# between two short waits, measured) is lost, and the next wait then runs its
|
||||
# whole timeout on a turn that already ended. The resend is a tagged DUPLICATE,
|
||||
# so the server skips the write and re-waits without retyping (§5.3).
|
||||
tmp=$(mktemp "${TMPDIR:-/tmp}/codeman-wait.XXXXXX") || return 1
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" > "$tmp" &
|
||||
bg=$!
|
||||
# The prompt's head with whitespace removed, matched literally (the "$head"
|
||||
# quoting inside ${c#...} keeps a * or ? in the prompt from acting as a glob).
|
||||
head=$(printf '%s' "$p" | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g" | head -c 24)
|
||||
while [ "$n" -lt 12 ] && [ ! -s "$tmp" ]; do # a non-empty file means the wait ended
|
||||
c=$(_composer_text "$sid")
|
||||
if [ "$c" = '?' ]; then
|
||||
[ "$n" -eq 0 ] || break # unreadable pane: one Enter, then trust it
|
||||
elif [ -z "$head" ] || [ "${c#"$head"}" = "$c" ]; then
|
||||
break # composer empty (taken) or holding other text
|
||||
fi
|
||||
n=$((n+1))
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
i=0; while [ "$i" -lt 10 ] && [ ! -s "$tmp" ]; do sleep 1; i=$((i+1)); done
|
||||
done
|
||||
wait "$bg"
|
||||
# The duplicate reports `delivered:false` -- truthfully, but about the wrong send.
|
||||
# The first one delivered, so carry that forward, or §1's cleanup reads a completed
|
||||
# turn as an undelivered one and keeps a finished worker forever.
|
||||
r=$(jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end' < "$tmp")
|
||||
rm -f "$tmp"
|
||||
fi
|
||||
printf '%s\n' "$r"
|
||||
}
|
||||
@@ -325,10 +372,10 @@ last_text() {
|
||||
# The stamp is the LAST line on purpose (a truncated write leaves it unset) and is kept
|
||||
# bare on purpose: the write condition above anchors on it with $, so an inline comment
|
||||
# here would fail that match and rewrite this file on every single bootstrap.
|
||||
CODEMAN_PREAMBLE=1.22.0
|
||||
CODEMAN_PREAMBLE=1.30.1
|
||||
PREAMBLE
|
||||
)
|
||||
. "$PRE"; [ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble at $PRE is stale or truncated: rm it and re-run this block"; exit 1; }
|
||||
. "$PRE"; [ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble at $PRE is stale or truncated: rm it and re-run this block"; exit 1; }
|
||||
```
|
||||
|
||||
Every later Bash call that touches the API starts with the same two loader lines from
|
||||
@@ -379,7 +426,7 @@ and no per-call body to hand-build.
|
||||
|
||||
```bash
|
||||
. "${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh" 2>/dev/null # §0 loader
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
N=(alpha beta) # INVENT one fresh case name per worker; never list cases first
|
||||
# (a name may carry a mode: `beta:deepseek`, see below)
|
||||
T=('reply with one line: the absolute path of your working directory'
|
||||
@@ -440,9 +487,11 @@ Four things this block leans on, each one link away, no detour needed to run it:
|
||||
skill: §5.1. Those workspaces do get hooks now, unless the operator disabled it.
|
||||
- `sendwait` supplies the `\r`, picks a fresh `seq`, and self-heals a stranded Enter.
|
||||
A prompt without the `\r` is never submitted (§3), a reused `seq` is silently
|
||||
swallowed as an already-applied duplicate, and an Enter eaten by an Ink repaint
|
||||
strands the prompt on the composer until a bare `\r` follows: all three are reasons
|
||||
to let `sendwait` build the call rather than hand-rolling it.
|
||||
swallowed as an already-applied duplicate, and a lost Enter strands the prompt on the
|
||||
composer until a bare `\r` follows: Claude Code 2.1.277 and later ignore Enter for the
|
||||
first 30 to 50 seconds after the composer paints while still taking the text, so
|
||||
`sendwait` reads the composer and keeps pressing Enter until the prompt has left it.
|
||||
All three are reasons to let `sendwait` build the call rather than hand-rolling it.
|
||||
- Each `sendwait` costs that worker one billed turn, as does every prompt you send it.
|
||||
- Deleting the sessions does **not** remove the case directories. They are marked as
|
||||
agent-created, so `GET /api/v1/cases/agent-created` lists them for cleanup: §5.14.
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# ---- Codeman agent preamble 1.22.0 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
# ---- Codeman agent preamble 1.30.1 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
API="${CODEMAN_API_URL:?CODEMAN_API_URL not set; refusing to guess}"
|
||||
SELF="${CODEMAN_SESSION_ID:?CODEMAN_SESSION_ID not set}"
|
||||
# Credentials, cheapest first. Your session has usually INHERITED the server's
|
||||
@@ -72,6 +72,27 @@ _trust_key() { # <sid> -> "confirm" | "move" | "" (nothing safe to press)
|
||||
| tr -d ' \t' | grep -i '❯[0-9.]*\(yes,itrustthisfolder\|no,exit\)' | tail -1 \
|
||||
| sed -e 's/.*[Yy]es,.*/confirm/' -e 's/.*[Nn]o,.*/move/'
|
||||
}
|
||||
# ---- the composer: is the prompt still sitting there, unsent? ----
|
||||
# ⚠️ Claude Code 2.1.277 (auto-installed 2026-09-18) takes typed text the moment the
|
||||
# composer paints but IGNORES Enter for the first 30-50 seconds after it: the \r that
|
||||
# Codeman sends 50 ms after the text and a lone nudge at 20 s both leave the prompt
|
||||
# stranded, with `0 tokens`, while the wait burns its whole timeout. Measured through
|
||||
# this very route: Enter at 28 s stranded, Enter at 51 s submitted. So sendwait READS
|
||||
# the composer and keeps pressing Enter while the prompt is still there.
|
||||
_composer_text() { # <sid> -> the composer's text with ALL whitespace removed: "" once
|
||||
# the prompt was taken, "?" when the pane shows no composer at all. The composer is
|
||||
# the LAST `❯` line: Claude Code echoes a submitted prompt with the same glyph higher
|
||||
# up in the transcript, so only the last one says whether the text was taken.
|
||||
local t
|
||||
t=$("${CURL[@]}" -G "$API/api/v1/sessions/$1/terminal" --data-urlencode 'full=1' \
|
||||
| jq -r '.data.terminalBuffer // empty' \
|
||||
| sed -e "s/$(printf '\033')\[[0-9;?]*[a-zA-Z]//g" -e "s/$(printf '\033')[()][AB0]//g" \
|
||||
| tr -d '\r' | grep -a '^[[:space:]]*❯' | tail -1)
|
||||
[ -n "$t" ] || { printf '?'; return 0; }
|
||||
# Claude Code draws a NO-BREAK SPACE (U+00A0) after the glyph, which [:space:] does
|
||||
# not cover, so it is stripped by its bytes, portably (BSD sed has no \xHH).
|
||||
printf '%s' "$t" | sed 's/^[[:space:]]*❯//' | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g"
|
||||
}
|
||||
_accept_trust() { # <sid> -> 0 once it has answered the dialog, 1 if it could not
|
||||
local sid="$1" k i=1
|
||||
while [ "$i" -le 6 ]; do
|
||||
@@ -184,12 +205,16 @@ spawn_workers() {
|
||||
# worker a silent no-op that still "succeeds" and reports the previous turn's state.
|
||||
# Pass seq explicitly for exactly one reason: resending a possibly-delivered frame as a
|
||||
# deliberate duplicate, at the SAME number (§5.3).
|
||||
# Delivery is SELF-HEALING: an Ink repaint occasionally eats the Enter, leaving the
|
||||
# typed prompt stranded on the composer while a long wait runs its whole timeout
|
||||
# (observed live). So the first wait is short; on its timeout a bare \r goes out (the
|
||||
# missing Enter when the prompt is stranded, a no-op when the turn is genuinely
|
||||
# running), then the ORIGINAL frame is resent unchanged, which the server takes as a
|
||||
# tagged duplicate: it re-waits without retyping (§5.3). Trustworthy for a worker
|
||||
# Delivery is SELF-HEALING: the Enter can be lost (an Ink repaint eats it, and Claude
|
||||
# Code 2.1.277+ ignores it outright for the first 30-50 s after the composer paints),
|
||||
# leaving the typed prompt stranded on the composer while a long wait runs its whole
|
||||
# timeout (observed live, twelve reviews in a row). So the first wait is short; on its
|
||||
# timeout the ORIGINAL frame is resent unchanged as a long re-wait (a tagged duplicate:
|
||||
# the server re-waits without retyping, §5.3) and kept open in the background, while
|
||||
# the composer is READ (_composer_text) and, as long as the prompt is still sitting
|
||||
# there, a bare \r goes out about every ten seconds, up to twelve times. An empty
|
||||
# composer ends the loop, so a prompt that was taken is never nudged again, and the
|
||||
# wait that was open the whole time is what reports the turn's end. Trustworthy for a worker
|
||||
# spawn_worker handed back -- claude (hooks vetted) or deepseek (status bridge) --
|
||||
# and for those only. Hook-less workspaces and the other modes resolve on flapping
|
||||
# idle: markers instead (§5.5). ⚠️ A dsh worker running a profile that does not
|
||||
@@ -197,7 +222,7 @@ spawn_workers() {
|
||||
# it accepts the send and then burns both waits. One timeout on a dsh worker whose
|
||||
# pane clearly finished means that profile, so switch that worker to markers.
|
||||
sendwait() {
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r c head n=0 tmp bg i
|
||||
# `wait:"stop,exit"`, never the `wait:true` default set: that set also carries
|
||||
# `idle`, which is INFERRED from output stabilization and flaps mid-turn. On a
|
||||
# dsh worker whose TUI repaints rarely the session reads `idle` while the model
|
||||
@@ -212,16 +237,38 @@ sendwait() {
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$body")
|
||||
if jq -e '.data.delivered and .data.wait.timedOut' <<<"$r" >/dev/null 2>&1; then
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
# The resend is a tagged DUPLICATE, so the server skips the write and reports
|
||||
# `delivered:false` for it -- truthfully, but about the wrong send. The first
|
||||
# one delivered, so carry that forward, or §1's cleanup reads a completed turn
|
||||
# as an undelivered one and keeps a finished worker forever.
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" \
|
||||
| jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end')
|
||||
# ⚠️ The long re-wait is registered FIRST and stays open for the rest of this call,
|
||||
# in the background, while the Enter loop below works the composer. Signals have
|
||||
# no history: a `stop` that fires while no wait is open (during a composer read
|
||||
# between two short waits, measured) is lost, and the next wait then runs its
|
||||
# whole timeout on a turn that already ended. The resend is a tagged DUPLICATE,
|
||||
# so the server skips the write and re-waits without retyping (§5.3).
|
||||
tmp=$(mktemp "${TMPDIR:-/tmp}/codeman-wait.XXXXXX") || return 1
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" > "$tmp" &
|
||||
bg=$!
|
||||
# The prompt's head with whitespace removed, matched literally (the "$head"
|
||||
# quoting inside ${c#...} keeps a * or ? in the prompt from acting as a glob).
|
||||
head=$(printf '%s' "$p" | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g" | head -c 24)
|
||||
while [ "$n" -lt 12 ] && [ ! -s "$tmp" ]; do # a non-empty file means the wait ended
|
||||
c=$(_composer_text "$sid")
|
||||
if [ "$c" = '?' ]; then
|
||||
[ "$n" -eq 0 ] || break # unreadable pane: one Enter, then trust it
|
||||
elif [ -z "$head" ] || [ "${c#"$head"}" = "$c" ]; then
|
||||
break # composer empty (taken) or holding other text
|
||||
fi
|
||||
n=$((n+1))
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
i=0; while [ "$i" -lt 10 ] && [ ! -s "$tmp" ]; do sleep 1; i=$((i+1)); done
|
||||
done
|
||||
wait "$bg"
|
||||
# The duplicate reports `delivered:false` -- truthfully, but about the wrong send.
|
||||
# The first one delivered, so carry that forward, or §1's cleanup reads a completed
|
||||
# turn as an undelivered one and keeps a finished worker forever.
|
||||
r=$(jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end' < "$tmp")
|
||||
rm -f "$tmp"
|
||||
fi
|
||||
printf '%s\n' "$r"
|
||||
}
|
||||
@@ -247,4 +294,4 @@ last_text() {
|
||||
# The stamp is the LAST line on purpose (a truncated write leaves it unset) and is kept
|
||||
# bare on purpose: the write condition above anchors on it with $, so an inline comment
|
||||
# here would fail that match and rewrite this file on every single bootstrap.
|
||||
CODEMAN_PREAMBLE=1.22.0
|
||||
CODEMAN_PREAMBLE=1.30.1
|
||||
|
||||
@@ -340,7 +340,7 @@ ESC=$(printf '\033')
|
||||
### Starting a worker
|
||||
|
||||
`POST /api/v1/quick-start` body (all optional):
|
||||
`{"caseName":"worker-1","mode":"claude","sessionName":"w9-worker","effort":"high"}`
|
||||
`{"caseName":"worker-1","mode":"claude","sessionName":"auth-worker","effort":"high"}`
|
||||
, `mode` ∈ `claude|shell|opencode|codex|gemini|antigravity|pi|grok|deepseek|omp`; response is
|
||||
`.data.{sessionId, caseName, casePath}`. Creates the case directory (a real directory
|
||||
on the user's disk) if missing, do not retry it in a loop, and remember the name.
|
||||
|
||||
@@ -101,11 +101,16 @@ the case name, read it from the listing.
|
||||
From Codeman 1.16 a LOCAL claude spawn passes `--name <session name>` when the local
|
||||
CLI is 2.1.224+ (`buildNameCliArgs`, `session-cli-builder.ts:97-101`, wired in at
|
||||
`tmux-manager.ts:797`), so a worker's peer name usually IS its Codeman session name
|
||||
(verified live: quick-start with `sessionName: "w9-msgtest"` listed as `w9-msgtest`,
|
||||
and its messages arrive tagged `from-name="w9-msgtest"`; a derived-name worker's
|
||||
(verified live: a quick-start `sessionName` is listed as that exact peer name, and
|
||||
the worker's messages arrive tagged `from-name="<that name>"`; a derived-name worker's
|
||||
messages carry no `from-name`). Name your workers: a quick-start WITHOUT
|
||||
`sessionName` leaves the Codeman name empty, so there is nothing to pass and the
|
||||
peer name stays derived. The flag is fail-closed (older/unknown CLI omits it, because an
|
||||
peer name stays derived. ⚠️ Give them a DESCRIPTIVE name: only a name the user chose
|
||||
is pinned (`Session.cliPinnedName`), because `--name` is also the conversation's
|
||||
`/resume` title and terminal title and suppresses Claude's own generated title. A
|
||||
placeholder-shaped name (`w9-msgtest`, anything matching `isGeneratedSessionName`)
|
||||
and an auto name are NOT passed, so such a worker's peer name is derived; use
|
||||
`msgtest-worker` rather than `w9-msgtest`. The flag is fail-closed (older/unknown CLI omits it, because an
|
||||
unknown flag aborts startup and would kill every spawn) and allowlist-sanitized (a name of
|
||||
only unsafe characters is dropped), and the docker/remote builders never see it at all
|
||||
(`tmux-manager.ts:782-789`), which is why the `tmux` column stays the canonical join key
|
||||
@@ -196,8 +201,8 @@ idle:
|
||||
The contract an orchestrator follows for any fleet of two or more messaging workers.
|
||||
Every topology in the next section is this protocol plus a wiring diagram.
|
||||
|
||||
1. **Spawn with a name, and confirm hooks.** Use `quick-start` with `sessionName` (the
|
||||
`--name` gate above). Session create installs the hooks block into the workspace
|
||||
1. **Spawn with a name, and confirm hooks.** Use `quick-start` with a descriptive,
|
||||
non-`w<N>-` `sessionName` (the `--name` gate above). Session create installs the hooks block into the workspace
|
||||
whatever kind it is, so a linked case and a raw `POST /api/sessions` path both get
|
||||
`stop`/`blocked` by default. ⚠️ Not unconditionally: the operator can turn
|
||||
`workspaceHooksEnabled` off, remote SSH sessions never get hooks, and a session from
|
||||
|
||||
@@ -21,7 +21,7 @@ by sourcing the preamble file the §0 bootstrap wrote, and checking its version
|
||||
|
||||
```bash
|
||||
. "${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh" 2>/dev/null
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble missing or stale; re-run the §0 bootstrap"; exit 1; }
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble missing or stale; re-run the §0 bootstrap"; exit 1; }
|
||||
```
|
||||
|
||||
Do **not** re-paste the preamble body into each call. Sourcing it is what retires the
|
||||
|
||||
@@ -692,8 +692,9 @@ The shape, each step verified live (probes, failure modes and safety detail in
|
||||
[§5.2](#52-readiness)).
|
||||
2. `ListAgents`: find the worker's row by its `tmux codeman-<first 8 of session id>`
|
||||
column; the row's `name [ref]` is the address. On Codeman 1.16+ with claude
|
||||
2.1.224+ a worker's peer name is its Codeman session name, so pass `sessionName`
|
||||
in quick-start to pick it; older setups list a name derived from the case folder.
|
||||
2.1.224+ a worker's peer name is its Codeman session name, so pass a DESCRIPTIVE
|
||||
`sessionName` in quick-start to pick it (a `w<N>-` placeholder-shaped name is not
|
||||
pinned, so it lists derived); older setups list a name derived from the case folder.
|
||||
No row = messaging is off for that worker (it is feature-flagged even on matching
|
||||
CLI versions, observed live): fall back to the HTTP recipes without complaint.
|
||||
3. `SendMessage` the task; first contact must use the `name [ref]` form copied from
|
||||
|
||||
@@ -146,10 +146,44 @@ console.log('\n[build] content-hash cache busting');
|
||||
html = html.replaceAll(`"${original}"`, `"${hashed}"`);
|
||||
}
|
||||
writeFileSync(join(distPublic, 'index.html'), html);
|
||||
|
||||
// Rewrite sw.js from the SAME manifest that just renamed the files.
|
||||
//
|
||||
// The service worker's precache list used to be maintained by hand with the
|
||||
// pre-hash names, so after this step every entry in it pointed at a file that
|
||||
// no longer existed and `cache.add(...).catch(() => {})` hid it. Deriving it
|
||||
// here is the only way the two cannot drift.
|
||||
//
|
||||
// The cache key gets the build hash for the same reason: `activate` deletes
|
||||
// every cache that is not the current one, so a constant key meant that
|
||||
// cleanup never ran and hashed assets from every past release piled up.
|
||||
const swPath = join(distPublic, 'sw.js');
|
||||
let sw = readFileSync(swPath, 'utf8');
|
||||
const hashedAssets = Object.values(manifest);
|
||||
const buildId = createHash('md5').update(hashedAssets.join('|')).digest('hex').slice(0, 12);
|
||||
// Rewrite the two declarations. Anchored on the full `const … = …;` text so
|
||||
// each pattern occurs exactly once and cannot collide with prose in sw.js's
|
||||
// own comments — an earlier cut used bare `__BUILD_ID__` sentinels and the
|
||||
// first match landed in the comment that documented them, leaving the real
|
||||
// constant untouched and still producing a plausible-looking cache key.
|
||||
const swEdits = [
|
||||
["const BUILD_ID = 'dev';", `const BUILD_ID = '${buildId}';`],
|
||||
['const HASHED_ASSETS = [];', `const HASHED_ASSETS = [${hashedAssets.map((p) => JSON.stringify(p)).join(', ')}];`],
|
||||
];
|
||||
for (const [from, to] of swEdits) {
|
||||
const hits = sw.split(from).length - 1;
|
||||
if (hits !== 1) {
|
||||
throw new Error(`sw.js: expected exactly one \`${from}\`, found ${hits} — precache would ship stale`);
|
||||
}
|
||||
sw = sw.replace(from, to);
|
||||
}
|
||||
writeFileSync(swPath, sw);
|
||||
|
||||
console.log(' Hashed files:');
|
||||
for (const [orig, hashed] of Object.entries(manifest)) {
|
||||
console.log(` ${orig} -> ${hashed}`);
|
||||
}
|
||||
console.log(` sw.js: cache bucket codeman-${buildId}, ${hashedAssets.length} precached assets`);
|
||||
}
|
||||
|
||||
// 6. Compress with gzip + brotli
|
||||
|
||||
@@ -0,0 +1,185 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* Browser-test exclusion check.
|
||||
*
|
||||
* `npm run test:ci` must never try to drive a real browser: CI runners (and any
|
||||
* clean checkout) have no chromium, so such a file dies with
|
||||
* `browserType.launch: Executable doesn't exist` and takes the whole suite with
|
||||
* it. `config/vitest.ci.config.ts` therefore excludes every browser-driven test
|
||||
* via `BROWSER_TEST_GLOBS` in `config/test-suites.ts`. That list is maintained
|
||||
* BY HAND, and a new browser test simply does not appear in it unless someone
|
||||
* remembers. The omission is invisible on a developer machine that has run
|
||||
* `npx playwright install`, where the test passes, and only shows up on a clean
|
||||
* runner.
|
||||
*
|
||||
* Two deliberate design choices:
|
||||
*
|
||||
* 1. **Detection is by CONTENT, not filename.** Matching `*.browser.test.ts`
|
||||
* would miss the browser tests that predate that convention
|
||||
* (`inline-rename`, `opencode-resize`, `webgl-fallback`,
|
||||
* `terminal-copy-shortcut`, `codex-predictive-echo`). What actually makes a
|
||||
* file dangerous is importing a browser driver, so that is what is tested.
|
||||
* ⚠️ Only a DIRECT import is seen: a test that reaches playwright through a
|
||||
* helper module (e.g. `test/mobile/helpers/browser.ts`) is not detected, so
|
||||
* such a test still has to be added to `BROWSER_TEST_GLOBS` by hand.
|
||||
*
|
||||
* 2. **The exclusion side is answered by vitest itself**, via
|
||||
* `vitest list --filesOnly`, rather than by re-implementing glob matching
|
||||
* against the config's `exclude` array. Patterns there include `test/mobile/**`
|
||||
* and `perf-*`; a hand-rolled matcher that disagreed with vitest by even one
|
||||
* edge case would report a gap that does not exist, or miss one that does.
|
||||
* Asking the real resolver cannot drift from the real behaviour.
|
||||
*
|
||||
* The pure pieces are exported for test/check-browser-test-excludes.test.ts; the
|
||||
* check itself only runs when this file is executed directly.
|
||||
*/
|
||||
import { readdirSync, readFileSync } from 'node:fs';
|
||||
import { join, dirname, relative, sep, resolve } from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
import { execFileSync } from 'node:child_process';
|
||||
|
||||
const ROOT = join(dirname(fileURLToPath(import.meta.url)), '..');
|
||||
const CI_CONFIG = join('config', 'vitest.ci.config.ts');
|
||||
const SUITES_FILE = join('config', 'test-suites.ts');
|
||||
|
||||
/** Importing any one of these means the test needs a real browser binary. */
|
||||
const BROWSER_DRIVER =
|
||||
/\bfrom\s+['"](?:playwright|playwright-core|@playwright\/test|puppeteer|puppeteer-core)['"]|\b(?:require|import)\(\s*['"](?:playwright|playwright-core|@playwright\/test|puppeteer|puppeteer-core)['"]\s*\)/;
|
||||
|
||||
/** @param {string} source */
|
||||
export function importsBrowserDriver(source) {
|
||||
return BROWSER_DRIVER.test(source);
|
||||
}
|
||||
|
||||
/** @param {string} dir @returns {string[]} */
|
||||
function walk(dir) {
|
||||
const out = [];
|
||||
for (const entry of readdirSync(dir, { withFileTypes: true })) {
|
||||
const path = join(dir, entry.name);
|
||||
if (entry.isDirectory()) out.push(...walk(path));
|
||||
else if (entry.isFile() && entry.name.endsWith('.test.ts')) out.push(path);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Every `*.test.ts` under `<root>/test`, as sorted repo-relative POSIX paths (the form
|
||||
* `vitest list` prints).
|
||||
*
|
||||
* @param {string} root
|
||||
* @returns {string[]}
|
||||
*/
|
||||
export function findTestFiles(root) {
|
||||
return walk(join(root, 'test'))
|
||||
.map((file) => relative(root, file).split(sep).join('/'))
|
||||
.sort();
|
||||
}
|
||||
|
||||
/**
|
||||
* The subset of {@link findTestFiles} that imports a browser driver.
|
||||
*
|
||||
* @param {string} root
|
||||
* @returns {string[]}
|
||||
*/
|
||||
export function findBrowserTests(root) {
|
||||
return findTestFiles(root).filter((file) => importsBrowserDriver(readFileSync(join(root, file), 'utf8')));
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse `vitest list --filesOnly` output into a set of repo-relative paths. Stray
|
||||
* blank or decorative lines are ignored rather than assuming the format is pristine.
|
||||
*
|
||||
* @param {string} output
|
||||
* @returns {Set<string>}
|
||||
*/
|
||||
export function parseVitestFileList(output) {
|
||||
return new Set(
|
||||
output
|
||||
.split('\n')
|
||||
.map((line) => line.trim())
|
||||
.filter((line) => line.endsWith('.test.ts'))
|
||||
.map((line) => line.replace(/^\.\//, ''))
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the `vitest list` paths and the walked tree name at least one file in common.
|
||||
* False means the two sides are not speaking the same path format (absolute paths, backslashes
|
||||
* or a new prefix after a vitest upgrade), and then {@link findLeaks} would find nothing
|
||||
* against a perfectly non-empty listing.
|
||||
*
|
||||
* @param {Set<string>} ciFiles
|
||||
* @param {string[]} testFiles
|
||||
*/
|
||||
export function listingMatchesTree(ciFiles, testFiles) {
|
||||
return testFiles.some((file) => ciFiles.has(file));
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {string[]} browserTests
|
||||
* @param {Set<string>} ciFiles
|
||||
* @returns {string[]} browser-driven files that the CI config would still collect
|
||||
*/
|
||||
export function findLeaks(browserTests, ciFiles) {
|
||||
return browserTests.filter((file) => ciFiles.has(file));
|
||||
}
|
||||
|
||||
function main() {
|
||||
const testFiles = findTestFiles(ROOT);
|
||||
const browserTests = findBrowserTests(ROOT);
|
||||
|
||||
let collected;
|
||||
try {
|
||||
collected = execFileSync('npx', ['vitest', 'list', '--config', CI_CONFIG, '--filesOnly'], {
|
||||
cwd: ROOT,
|
||||
encoding: 'utf8',
|
||||
stdio: ['ignore', 'pipe', 'pipe'],
|
||||
});
|
||||
} catch (err) {
|
||||
console.error('✗ could not enumerate the CI test set via `vitest list`.');
|
||||
console.error(err.stderr ? err.stderr.toString() : String(err));
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const ciFiles = parseVitestFileList(collected);
|
||||
if (ciFiles.size === 0) {
|
||||
// An empty list would make every browser test look excluded: fail rather than pass vacuously.
|
||||
console.error('✗ `vitest list` reported no test files; refusing to pass on an empty CI set.');
|
||||
process.exit(1);
|
||||
}
|
||||
// Same vacuous pass, one step removed: a listing whose paths never match the tree. This guard,
|
||||
// not `vitest list --json`, is the answer to format drift: the JSON form prints absolute paths
|
||||
// that would need canonicalizing against ROOT (symlinked checkouts), and its shape can drift too.
|
||||
if (!listingMatchesTree(ciFiles, testFiles)) {
|
||||
const sample = [...ciFiles].slice(0, 3).join(', ');
|
||||
console.error(
|
||||
`✗ none of the ${ciFiles.size} paths \`vitest list\` reported (e.g. ${sample}) is one of the ${testFiles.length} test/**/*.test.ts files; its output format has probably changed.`
|
||||
);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const leaked = findLeaks(browserTests, ciFiles);
|
||||
|
||||
if (leaked.length > 0) {
|
||||
console.error(`✗ ${leaked.length} browser-driven test file(s) are NOT excluded from ${CI_CONFIG}:\n`);
|
||||
for (const file of leaked) console.error(` ${file}`);
|
||||
console.error(`
|
||||
These import a browser driver, so on a runner with no chromium they fail with
|
||||
"browserType.launch: Executable doesn't exist" and take the suite down. Add each
|
||||
to BROWSER_TEST_GLOBS in ${SUITES_FILE} (${CI_CONFIG} derives its excludes from
|
||||
it, and \`npm run test:browser\` its includes).
|
||||
|
||||
They may well pass on this machine; that is the trap. To reproduce a clean
|
||||
runner locally:
|
||||
PLAYWRIGHT_BROWSERS_PATH=\$(mktemp -d) PUPPETEER_CACHE_DIR=\$(mktemp -d) npm run test:ci`);
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
console.log(
|
||||
`✓ all ${browserTests.length} browser-driven test files are excluded from the CI suite (${ciFiles.size} files collected)`
|
||||
);
|
||||
}
|
||||
|
||||
if (process.argv[1] && resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
|
||||
main();
|
||||
}
|
||||
@@ -0,0 +1,253 @@
|
||||
/**
|
||||
* @fileoverview Git hook bodies + install policy, shared by scripts/postinstall.js and
|
||||
* pinned by test/git-hooks.test.ts.
|
||||
*
|
||||
* Why a pre-push hook: the static CI job (lockfile, typecheck, lint, format, frontend
|
||||
* syntax, ...) fails often on things a contributor could have caught locally in seconds,
|
||||
* and finding out after a push costs a full CI round-trip plus a fix-up commit. Running
|
||||
* the same checks before the push surfaces those failures in ~10-40s instead (12s on a fast
|
||||
* workstation, ~35s measured elsewhere; typecheck, format:check and lint dominate).
|
||||
*
|
||||
* Why pre-PUSH and not pre-commit: a commit is cheap and local, a push is what CI and
|
||||
* reviewers pick up. And why the STATIC tier only: the unit/integration suite takes
|
||||
* minutes, which nobody tolerates per push, so a hook that ran it would be bypassed
|
||||
* within a day. The checks below mirror the static CI job.
|
||||
*
|
||||
* ⚠️ The checks read the WORKING TREE, not the commits being pushed. So the hook skips
|
||||
* (with a one-line notice) whenever the two can differ: when HEAD is not the commit being
|
||||
* pushed, and when `git status` shows uncommitted or untracked changes in a path a check
|
||||
* reads ({@link PRE_PUSH_WATCHED_PATHS}). In a checkout shared by several agent sessions
|
||||
* the second case is usually another session's WIP, which must not block this push.
|
||||
*
|
||||
* ⚠️ This installer is deliberately MARKER-OWNED, unlike the older pre-commit installer in
|
||||
* postinstall.js which overwrites whatever it finds. A developer's own pre-push hook must
|
||||
* survive `npm install`.
|
||||
*/
|
||||
|
||||
import { execFileSync } from 'node:child_process';
|
||||
import { chmodSync, existsSync, mkdirSync, readFileSync, realpathSync, writeFileSync } from 'node:fs';
|
||||
import { basename, dirname, join, resolve } from 'node:path';
|
||||
|
||||
/**
|
||||
* Ownership marker. ⚠️ Never bump the version suffix: ownership is matched on this exact
|
||||
* string, so a `v2` would read every installed `v1` hook as foreign and never refresh it.
|
||||
* A changed body still reaches installed hooks, because the refresh compares the whole file.
|
||||
*/
|
||||
export const PRE_PUSH_MARKER = '# codeman-managed-hook: pre-push v1';
|
||||
|
||||
/**
|
||||
* Checks that make up the fast tier, cheapest first so failures surface sooner. Each entry
|
||||
* is the argument list for `npm run`, and each is a step of the static job in
|
||||
* .github/workflows/ci.yml (test/git-hooks.test.ts pins that every script exists).
|
||||
*/
|
||||
export const PRE_PUSH_CHECKS = [
|
||||
['check:lockfile'],
|
||||
['generate:cli-catalog', '--', '--check'],
|
||||
['check:browser-excludes'],
|
||||
['check:frontend-syntax'],
|
||||
['format:check'],
|
||||
['lint'],
|
||||
['typecheck'],
|
||||
];
|
||||
|
||||
/**
|
||||
* Paths whose uncommitted state would leak into a check, so a dirty one makes the hook skip.
|
||||
* Derived from what each check reads: src/ (format:check, lint, typecheck,
|
||||
* check:frontend-syntax), config/ (eslint + vitest configs, test-suites.ts, the CLI
|
||||
* catalogue), scripts/ (every check is a script there, and typecheck's second pass compiles
|
||||
* one), test/ (check:browser-excludes scans it and runs `vitest list` over it),
|
||||
* package.json + package-lock.json (check:lockfile), install.sh (generate:cli-catalog
|
||||
* --check diffs its generated block), tsconfig.json (typecheck, and
|
||||
* config/tsconfig.scripts.json extends it) and .prettierignore + .editorconfig
|
||||
* (format:check; the Prettier CLI honours .editorconfig by default).
|
||||
*/
|
||||
export const PRE_PUSH_WATCHED_PATHS = [
|
||||
'src',
|
||||
'config',
|
||||
'scripts',
|
||||
'test',
|
||||
'package.json',
|
||||
'package-lock.json',
|
||||
'install.sh',
|
||||
'tsconfig.json',
|
||||
'.prettierignore',
|
||||
'.editorconfig',
|
||||
];
|
||||
|
||||
/**
|
||||
* Render the pre-push hook script.
|
||||
*
|
||||
* POSIX sh, not bash: this ships to whatever shell the contributor's git uses.
|
||||
*/
|
||||
export function renderPrePushHook() {
|
||||
const runs = PRE_PUSH_CHECKS.map((args) => `run_check ${args.join(' ')}`).join('\n');
|
||||
const watched = PRE_PUSH_WATCHED_PATHS.join(' ');
|
||||
|
||||
return `#!/bin/sh
|
||||
${PRE_PUSH_MARKER}
|
||||
# Installed by scripts/postinstall.js. Edit scripts/git-hooks.mjs, not this file:
|
||||
# it is regenerated on npm install. Delete the marker line above to take ownership
|
||||
# and the installer will leave your version alone.
|
||||
#
|
||||
# Skip once: CODEMAN_SKIP_PREPUSH=1 git push
|
||||
# Skip always: remove this file.
|
||||
|
||||
[ "$CODEMAN_SKIP_PREPUSH" = "1" ] && exit 0
|
||||
|
||||
repo_root=$(git rev-parse --show-toplevel 2>/dev/null) || exit 0
|
||||
cd "$repo_root" || exit 0
|
||||
|
||||
# Nothing to check without dependencies (fresh clone, or a worktree that never ran
|
||||
# npm install). Warn rather than blocking the push on a setup detail.
|
||||
if [ ! -d node_modules ]; then
|
||||
echo "pre-push: node_modules missing, skipping checks (run 'npm install' to enable them)."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# GUI git clients and IDEs often run hooks with a minimal PATH that lacks an nvm or
|
||||
# Homebrew Node. Every check would then fail with "npm: not found", so skip instead.
|
||||
command -v npm >/dev/null 2>&1 || { echo "pre-push: npm not on PATH, skipping checks."; exit 0; }
|
||||
|
||||
# git feeds us "<localref> <localsha> <remoteref> <remotesha>" per ref. A deletion has an
|
||||
# all-zero local sha and no tree worth checking; if every ref is a deletion, skip.
|
||||
# The checks below read the working tree, so they only say something about a pushed commit
|
||||
# that IS the checked-out HEAD (tags are peeled to their commit first).
|
||||
head=$(git rev-parse -q --verify HEAD 2>/dev/null)
|
||||
has_content=0
|
||||
not_head=''
|
||||
while read -r localref localsha _remoteref _remotesha; do
|
||||
[ -z "$localsha" ] && continue
|
||||
case "$localsha" in
|
||||
0000000000000000000000000000000000000000) ;;
|
||||
*)
|
||||
has_content=1
|
||||
commit=$(git rev-parse -q --verify "$localsha^{commit}" 2>/dev/null)
|
||||
[ -n "$head" ] && [ "$commit" = "$head" ] || not_head="$localref"
|
||||
;;
|
||||
esac
|
||||
done
|
||||
[ "$has_content" = "0" ] && exit 0
|
||||
|
||||
if [ -n "$not_head" ]; then
|
||||
echo "pre-push: skipping static checks: $not_head is not the checked-out HEAD, and the checks read the working tree."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Uncommitted or untracked changes in a path a check reads would be judged instead of the
|
||||
# pushed commit. In a checkout shared by several sessions that is usually someone else's WIP.
|
||||
if [ -n "$(git --no-optional-locks status --porcelain -- ${watched} 2>/dev/null)" ]; then
|
||||
echo "pre-push: skipping static checks: uncommitted changes under ${watched} would be checked instead of the pushed commit."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
log=$(mktemp "\${TMPDIR:-/tmp}/codeman-prepush.XXXXXX") || exit 0
|
||||
trap 'rm -f "$log"' EXIT
|
||||
|
||||
failed=''
|
||||
run_check() {
|
||||
if ! npm run --silent "$@" >"$log" 2>&1; then
|
||||
echo ""
|
||||
echo "pre-push: FAILED npm run $*"
|
||||
tail -n 25 "$log"
|
||||
failed="$failed $1"
|
||||
fi
|
||||
}
|
||||
|
||||
echo "pre-push: running static checks (~10-40s)..."
|
||||
${runs}
|
||||
|
||||
if [ -n "$failed" ]; then
|
||||
echo ""
|
||||
echo "pre-push: blocked by:$failed"
|
||||
echo "Fix, or push anyway with: CODEMAN_SKIP_PREPUSH=1 git push"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "pre-push: static checks passed."
|
||||
exit 0
|
||||
`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Decide what to do with an existing hook file.
|
||||
*
|
||||
* @param {{ existing: string | null | undefined, next: string }} args
|
||||
* @returns {'write' | 'up-to-date' | 'skip-foreign'}
|
||||
*/
|
||||
export function planHookInstall({ existing, next }) {
|
||||
if (existing === null || existing === undefined || existing.trim() === '') return 'write';
|
||||
if (!existing.includes(PRE_PUSH_MARKER)) return 'skip-foreign';
|
||||
return existing === next ? 'up-to-date' : 'write';
|
||||
}
|
||||
|
||||
/** @param {string} cwd @param {string[]} args */
|
||||
function git(cwd, args) {
|
||||
return execFileSync('git', args, { cwd, encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'] }).trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* realpath() that tolerates a missing leaf: a fresh `.git` may have no `hooks/` yet, so
|
||||
* canonicalize the parent and re-append the name. Throws if the parent is missing too.
|
||||
*
|
||||
* @param {string} path
|
||||
*/
|
||||
function canonicalPath(path) {
|
||||
return existsSync(path) ? realpathSync(path) : join(realpathSync(dirname(path)), basename(path));
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the hooks directory for the checkout rooted at `repoRoot`, or null when there
|
||||
* is nothing to install into.
|
||||
*
|
||||
* Asks git (`--git-path hooks`) rather than assuming `<root>/.git/hooks`: in a worktree
|
||||
* `.git` is a FILE pointing at the parent repo, so the hooks live under
|
||||
* `--git-common-dir`.
|
||||
*
|
||||
* ⚠️ Returns a directory ONLY when it is this repository's own `<git-common-dir>/hooks`.
|
||||
* `--git-path hooks` also reports `core.hooksPath`, and that setting is often GLOBAL (a
|
||||
* shared hooks directory used by every repo on the machine); installing there would
|
||||
* overwrite the user's own hooks and run Codeman's checks on unrelated repos. A
|
||||
* `core.hooksPath` that points back at the repo's own hooks dir still resolves, because
|
||||
* the comparison is on canonical paths rather than on whether the setting exists.
|
||||
*
|
||||
* Also returns null unless `repoRoot` is itself the top of a work tree. Without that guard,
|
||||
* a copy of this package sitting inside SOMEONE ELSE's repository (e.g. under their
|
||||
* node_modules) would resolve to their hooks directory and install Codeman's hook there.
|
||||
*
|
||||
* @param {string} repoRoot
|
||||
* @returns {string | null}
|
||||
*/
|
||||
export function resolveGitHooksDir(repoRoot) {
|
||||
try {
|
||||
const top = git(repoRoot, ['rev-parse', '--show-toplevel']);
|
||||
if (!top || realpathSync(top) !== realpathSync(repoRoot)) return null;
|
||||
// Both are printed relative to the cwd (repoRoot) unless already absolute.
|
||||
const hooks = git(repoRoot, ['rev-parse', '--git-path', 'hooks']);
|
||||
const common = git(repoRoot, ['rev-parse', '--git-common-dir']);
|
||||
if (!hooks || !common) return null;
|
||||
const own = join(realpathSync(resolve(repoRoot, common)), 'hooks');
|
||||
return canonicalPath(resolve(repoRoot, hooks)) === own ? own : null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Install (or refresh) the managed pre-push hook in `hooksDir`, honouring
|
||||
* {@link planHookInstall}: a hook without the marker is never touched.
|
||||
*
|
||||
* @param {string} hooksDir
|
||||
* @returns {'write' | 'up-to-date' | 'skip-foreign'}
|
||||
*/
|
||||
export function installPrePushHook(hooksDir) {
|
||||
const path = join(hooksDir, 'pre-push');
|
||||
const next = renderPrePushHook();
|
||||
const existing = existsSync(path) ? readFileSync(path, 'utf8') : null;
|
||||
const action = planHookInstall({ existing, next });
|
||||
if (action === 'write') {
|
||||
mkdirSync(hooksDir, { recursive: true });
|
||||
writeFileSync(path, next, { mode: 0o755 });
|
||||
chmodSync(path, 0o755); // `mode` only applies when the file is created
|
||||
}
|
||||
return action;
|
||||
}
|
||||
@@ -55,9 +55,61 @@ export function agentImageNpmPackages(catalog) {
|
||||
return packages;
|
||||
}
|
||||
|
||||
/** The `--build-arg` pairs the agent image takes. PURE. */
|
||||
export function agentImageBuildArgPairs(catalog) {
|
||||
return [['CLI_NPM_PACKAGES', agentImageNpmPackages(catalog).join(' ')]];
|
||||
/**
|
||||
* Environment variable → agent.Dockerfile ARG for the optional git-host CLIs (gh, az).
|
||||
* ⚠️ Mirrored by `GIT_HOST_CLI_BUILD_ARGS` in `src/docker-hosts.ts`; the parity test pins them.
|
||||
*/
|
||||
export const GIT_HOST_CLI_BUILD_ARGS = [
|
||||
['CODEMAN_AGENT_IMAGE_INSTALL_GH', 'CODEMAN_INSTALL_GH'],
|
||||
['CODEMAN_AGENT_IMAGE_INSTALL_AZ', 'CODEMAN_INSTALL_AZ'],
|
||||
];
|
||||
|
||||
/**
|
||||
* Environment variable → Dockerfile ARG for the image's system Git identity.
|
||||
* ⚠️ Mirrored by `GIT_IDENTITY_BUILD_ARGS` in `src/docker-hosts.ts`; the parity test pins them.
|
||||
*/
|
||||
export const GIT_IDENTITY_BUILD_ARGS = [
|
||||
['CODEMAN_AGENT_IMAGE_GIT_USER_NAME', 'GIT_USER_NAME'],
|
||||
['CODEMAN_AGENT_IMAGE_GIT_USER_EMAIL', 'GIT_USER_EMAIL'],
|
||||
];
|
||||
|
||||
/**
|
||||
* The `--build-arg` pairs for the optional git-host CLIs. PURE. An unset or empty variable
|
||||
* contributes NOTHING, so the Dockerfile's own default (off) applies and the argv is the same
|
||||
* as before these existed; anything other than 0/1 is refused rather than guessed at.
|
||||
*/
|
||||
export function gitHostCliBuildArgPairs(env) {
|
||||
const pairs = [];
|
||||
for (const [envName, argName] of GIT_HOST_CLI_BUILD_ARGS) {
|
||||
const value = env[envName];
|
||||
if (value === undefined || value === '') continue;
|
||||
if (value !== '0' && value !== '1') {
|
||||
throw new Error(`${envName} must be 0 or 1, got ${JSON.stringify(value)}`);
|
||||
}
|
||||
pairs.push([argName, value]);
|
||||
}
|
||||
return pairs;
|
||||
}
|
||||
|
||||
/** The `--build-arg` pairs for Git identity, requiring either both values or neither. */
|
||||
export function gitIdentityBuildArgPairs(env) {
|
||||
const pairs = GIT_IDENTITY_BUILD_ARGS.map(([envName, argName]) => [argName, env[envName] ?? '']);
|
||||
const configured = pairs.filter(([, value]) => value !== '');
|
||||
if (configured.length === 0) return [];
|
||||
if (configured.length !== pairs.length) {
|
||||
const names = GIT_IDENTITY_BUILD_ARGS.map(([envName]) => envName).join(' and ');
|
||||
throw new Error(`${names} must both be set when configuring Git identity`);
|
||||
}
|
||||
return pairs;
|
||||
}
|
||||
|
||||
/** The `--build-arg` pairs the agent image takes. PURE given `env`. */
|
||||
export function agentImageBuildArgPairs(catalog, env = process.env) {
|
||||
return [
|
||||
['CLI_NPM_PACKAGES', agentImageNpmPackages(catalog).join(' ')],
|
||||
...gitHostCliBuildArgPairs(env),
|
||||
...gitIdentityBuildArgPairs(env),
|
||||
];
|
||||
}
|
||||
|
||||
/** Read the committed catalogue. IO. */
|
||||
|
||||
+16
-4
@@ -356,14 +356,17 @@ if (!isGlobalInstall) {
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// 5. Install git pre-commit hook (format check)
|
||||
// 5. Install git hooks (pre-commit format check, pre-push static checks)
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
if (!isGlobalInstall) {
|
||||
try {
|
||||
const { writeFileSync, mkdirSync } = await import('fs');
|
||||
const gitHooksDir = join(import.meta.dirname, '..', '.git', 'hooks');
|
||||
if (existsSync(join(import.meta.dirname, '..', '.git'))) {
|
||||
const { resolveGitHooksDir, installPrePushHook } = await import('./git-hooks.mjs');
|
||||
// Resolved through git, not `../.git/hooks`: in a worktree `.git` is a file.
|
||||
// null when this directory is not the top of a git checkout.
|
||||
const gitHooksDir = resolveGitHooksDir(join(import.meta.dirname, '..'));
|
||||
if (gitHooksDir) {
|
||||
mkdirSync(gitHooksDir, { recursive: true });
|
||||
const hook = `#!/bin/bash
|
||||
# Auto-installed by postinstall — prevents CI format failures
|
||||
@@ -379,9 +382,18 @@ fi
|
||||
const hookPath = join(gitHooksDir, 'pre-commit');
|
||||
writeFileSync(hookPath, hook, { mode: 0o755 });
|
||||
console.log(colors.green('✓ Git pre-commit hook installed (prettier check)'));
|
||||
|
||||
// Unlike the pre-commit hook above, this one is marker-owned: a pre-push
|
||||
// hook the developer wrote themselves is left alone.
|
||||
const action = installPrePushHook(gitHooksDir);
|
||||
if (action === 'write') {
|
||||
console.log(colors.green('✓ Git pre-push hook installed') + colors.dim(' (static CI checks, ~10-40s)'));
|
||||
} else if (action === 'skip-foreign') {
|
||||
console.log(colors.dim(' Existing pre-push hook left untouched (not Codeman-managed)'));
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
// Non-critical — git hook is a convenience
|
||||
// Non-critical — git hooks are a convenience
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+20
-1
@@ -74,6 +74,15 @@ echo "[self-update] $(date) start tag=$TAG supervisor=$SUPERVISOR repo=$REPO"
|
||||
export PATH="$(dirname "$NODE"):$HOME/.local/bin:$HOME/.npm-global/bin:/usr/local/bin:/opt/homebrew/bin:$PATH"
|
||||
export GIT_TERMINAL_PROMPT=0
|
||||
|
||||
# --node is the server's process.execPath, a VERSIONED path (Homebrew resolves it
|
||||
# into Cellar/node/<ver>/). A `brew upgrade node` under a long-running server
|
||||
# deletes it, and every status write then failed, so the status stayed "queued"
|
||||
# forever. Fall back to whatever node is on PATH.
|
||||
if [ ! -x "$NODE" ]; then
|
||||
echo "[self-update] WARN: $NODE is not executable, falling back to node on PATH"
|
||||
NODE="$(command -v node || echo node)"
|
||||
fi
|
||||
|
||||
TO_VERSION="${TAG##*@}" # codeman@0.9.4 → 0.9.4 (tag is validated upstream)
|
||||
STASH_REF=""
|
||||
MANUAL_CMD=""
|
||||
@@ -276,7 +285,17 @@ case "$SUPERVISOR" in
|
||||
# domain needs root, but we don't need it — kill the server and launchd
|
||||
# respawns it on the new dist/ within ThrottleInterval seconds.
|
||||
if [[ -n "$SERVER_PID" ]] && kill "$SERVER_PID" 2>/dev/null; then
|
||||
: # respawn is launchd's job from here
|
||||
# Respawn is launchd's job, but only once the old process EXITS. A graceful
|
||||
# shutdown that hangs leaves the port closed and the service down, so
|
||||
# escalate to SIGKILL (tmux sessions live outside the server and survive).
|
||||
for _ in $(seq 1 30); do
|
||||
kill -0 "$SERVER_PID" 2>/dev/null || break
|
||||
sleep 1
|
||||
done
|
||||
if kill -0 "$SERVER_PID" 2>/dev/null; then
|
||||
echo "[self-update] server pid $SERVER_PID still alive 30s after SIGTERM, sending SIGKILL"
|
||||
kill -9 "$SERVER_PID" 2>/dev/null || true
|
||||
fi
|
||||
else
|
||||
MANUAL_CMD="sudo launchctl kickstart -k system/com.codeman.web"
|
||||
write_status "completed-needs-manual-restart" "Update staged — restart Codeman to apply v$TO_VERSION."
|
||||
|
||||
+75
-26
@@ -47,7 +47,7 @@ later call opens with, and your first REAL call performs them anyway:
|
||||
|
||||
```bash
|
||||
. "${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh" 2>/dev/null
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
```
|
||||
|
||||
⚠️ **Never spend a Bash call on this check alone.** §1's block opens with this same
|
||||
@@ -75,8 +75,8 @@ PRE="${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh"
|
||||
mkdir -p "$(dirname "$PRE")"
|
||||
# Rewrite unless the file already ends with THIS version's stamp, so a stale or a
|
||||
# half-written file self-heals here instead of costing you a round trip to rm it.
|
||||
grep -qs '^CODEMAN_PREAMBLE=1.22.0$' "$PRE" || (umask 077; cat > "$PRE" <<'PREAMBLE'
|
||||
# ---- Codeman agent preamble 1.22.0 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
grep -qs '^CODEMAN_PREAMBLE=1.30.1$' "$PRE" || (umask 077; cat > "$PRE" <<'PREAMBLE'
|
||||
# ---- Codeman agent preamble 1.30.1 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
API="${CODEMAN_API_URL:?CODEMAN_API_URL not set; refusing to guess}"
|
||||
SELF="${CODEMAN_SESSION_ID:?CODEMAN_SESSION_ID not set}"
|
||||
# Credentials, cheapest first. Your session has usually INHERITED the server's
|
||||
@@ -150,6 +150,27 @@ _trust_key() { # <sid> -> "confirm" | "move" | "" (nothing safe to press)
|
||||
| tr -d ' \t' | grep -i '❯[0-9.]*\(yes,itrustthisfolder\|no,exit\)' | tail -1 \
|
||||
| sed -e 's/.*[Yy]es,.*/confirm/' -e 's/.*[Nn]o,.*/move/'
|
||||
}
|
||||
# ---- the composer: is the prompt still sitting there, unsent? ----
|
||||
# ⚠️ Claude Code 2.1.277 (auto-installed 2026-09-18) takes typed text the moment the
|
||||
# composer paints but IGNORES Enter for the first 30-50 seconds after it: the \r that
|
||||
# Codeman sends 50 ms after the text and a lone nudge at 20 s both leave the prompt
|
||||
# stranded, with `0 tokens`, while the wait burns its whole timeout. Measured through
|
||||
# this very route: Enter at 28 s stranded, Enter at 51 s submitted. So sendwait READS
|
||||
# the composer and keeps pressing Enter while the prompt is still there.
|
||||
_composer_text() { # <sid> -> the composer's text with ALL whitespace removed: "" once
|
||||
# the prompt was taken, "?" when the pane shows no composer at all. The composer is
|
||||
# the LAST `❯` line: Claude Code echoes a submitted prompt with the same glyph higher
|
||||
# up in the transcript, so only the last one says whether the text was taken.
|
||||
local t
|
||||
t=$("${CURL[@]}" -G "$API/api/v1/sessions/$1/terminal" --data-urlencode 'full=1' \
|
||||
| jq -r '.data.terminalBuffer // empty' \
|
||||
| sed -e "s/$(printf '\033')\[[0-9;?]*[a-zA-Z]//g" -e "s/$(printf '\033')[()][AB0]//g" \
|
||||
| tr -d '\r' | grep -a '^[[:space:]]*❯' | tail -1)
|
||||
[ -n "$t" ] || { printf '?'; return 0; }
|
||||
# Claude Code draws a NO-BREAK SPACE (U+00A0) after the glyph, which [:space:] does
|
||||
# not cover, so it is stripped by its bytes, portably (BSD sed has no \xHH).
|
||||
printf '%s' "$t" | sed 's/^[[:space:]]*❯//' | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g"
|
||||
}
|
||||
_accept_trust() { # <sid> -> 0 once it has answered the dialog, 1 if it could not
|
||||
local sid="$1" k i=1
|
||||
while [ "$i" -le 6 ]; do
|
||||
@@ -262,12 +283,16 @@ spawn_workers() {
|
||||
# worker a silent no-op that still "succeeds" and reports the previous turn's state.
|
||||
# Pass seq explicitly for exactly one reason: resending a possibly-delivered frame as a
|
||||
# deliberate duplicate, at the SAME number (§5.3).
|
||||
# Delivery is SELF-HEALING: an Ink repaint occasionally eats the Enter, leaving the
|
||||
# typed prompt stranded on the composer while a long wait runs its whole timeout
|
||||
# (observed live). So the first wait is short; on its timeout a bare \r goes out (the
|
||||
# missing Enter when the prompt is stranded, a no-op when the turn is genuinely
|
||||
# running), then the ORIGINAL frame is resent unchanged, which the server takes as a
|
||||
# tagged duplicate: it re-waits without retyping (§5.3). Trustworthy for a worker
|
||||
# Delivery is SELF-HEALING: the Enter can be lost (an Ink repaint eats it, and Claude
|
||||
# Code 2.1.277+ ignores it outright for the first 30-50 s after the composer paints),
|
||||
# leaving the typed prompt stranded on the composer while a long wait runs its whole
|
||||
# timeout (observed live, twelve reviews in a row). So the first wait is short; on its
|
||||
# timeout the ORIGINAL frame is resent unchanged as a long re-wait (a tagged duplicate:
|
||||
# the server re-waits without retyping, §5.3) and kept open in the background, while
|
||||
# the composer is READ (_composer_text) and, as long as the prompt is still sitting
|
||||
# there, a bare \r goes out about every ten seconds, up to twelve times. An empty
|
||||
# composer ends the loop, so a prompt that was taken is never nudged again, and the
|
||||
# wait that was open the whole time is what reports the turn's end. Trustworthy for a worker
|
||||
# spawn_worker handed back -- claude (hooks vetted) or deepseek (status bridge) --
|
||||
# and for those only. Hook-less workspaces and the other modes resolve on flapping
|
||||
# idle: markers instead (§5.5). ⚠️ A dsh worker running a profile that does not
|
||||
@@ -275,7 +300,7 @@ spawn_workers() {
|
||||
# it accepts the send and then burns both waits. One timeout on a dsh worker whose
|
||||
# pane clearly finished means that profile, so switch that worker to markers.
|
||||
sendwait() {
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r c head n=0 tmp bg i
|
||||
# `wait:"stop,exit"`, never the `wait:true` default set: that set also carries
|
||||
# `idle`, which is INFERRED from output stabilization and flaps mid-turn. On a
|
||||
# dsh worker whose TUI repaints rarely the session reads `idle` while the model
|
||||
@@ -290,16 +315,38 @@ sendwait() {
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$body")
|
||||
if jq -e '.data.delivered and .data.wait.timedOut' <<<"$r" >/dev/null 2>&1; then
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
# The resend is a tagged DUPLICATE, so the server skips the write and reports
|
||||
# `delivered:false` for it -- truthfully, but about the wrong send. The first
|
||||
# one delivered, so carry that forward, or §1's cleanup reads a completed turn
|
||||
# as an undelivered one and keeps a finished worker forever.
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" \
|
||||
| jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end')
|
||||
# ⚠️ The long re-wait is registered FIRST and stays open for the rest of this call,
|
||||
# in the background, while the Enter loop below works the composer. Signals have
|
||||
# no history: a `stop` that fires while no wait is open (during a composer read
|
||||
# between two short waits, measured) is lost, and the next wait then runs its
|
||||
# whole timeout on a turn that already ended. The resend is a tagged DUPLICATE,
|
||||
# so the server skips the write and re-waits without retyping (§5.3).
|
||||
tmp=$(mktemp "${TMPDIR:-/tmp}/codeman-wait.XXXXXX") || return 1
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" > "$tmp" &
|
||||
bg=$!
|
||||
# The prompt's head with whitespace removed, matched literally (the "$head"
|
||||
# quoting inside ${c#...} keeps a * or ? in the prompt from acting as a glob).
|
||||
head=$(printf '%s' "$p" | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g" | head -c 24)
|
||||
while [ "$n" -lt 12 ] && [ ! -s "$tmp" ]; do # a non-empty file means the wait ended
|
||||
c=$(_composer_text "$sid")
|
||||
if [ "$c" = '?' ]; then
|
||||
[ "$n" -eq 0 ] || break # unreadable pane: one Enter, then trust it
|
||||
elif [ -z "$head" ] || [ "${c#"$head"}" = "$c" ]; then
|
||||
break # composer empty (taken) or holding other text
|
||||
fi
|
||||
n=$((n+1))
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
i=0; while [ "$i" -lt 10 ] && [ ! -s "$tmp" ]; do sleep 1; i=$((i+1)); done
|
||||
done
|
||||
wait "$bg"
|
||||
# The duplicate reports `delivered:false` -- truthfully, but about the wrong send.
|
||||
# The first one delivered, so carry that forward, or §1's cleanup reads a completed
|
||||
# turn as an undelivered one and keeps a finished worker forever.
|
||||
r=$(jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end' < "$tmp")
|
||||
rm -f "$tmp"
|
||||
fi
|
||||
printf '%s\n' "$r"
|
||||
}
|
||||
@@ -325,10 +372,10 @@ last_text() {
|
||||
# The stamp is the LAST line on purpose (a truncated write leaves it unset) and is kept
|
||||
# bare on purpose: the write condition above anchors on it with $, so an inline comment
|
||||
# here would fail that match and rewrite this file on every single bootstrap.
|
||||
CODEMAN_PREAMBLE=1.22.0
|
||||
CODEMAN_PREAMBLE=1.30.1
|
||||
PREAMBLE
|
||||
)
|
||||
. "$PRE"; [ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble at $PRE is stale or truncated: rm it and re-run this block"; exit 1; }
|
||||
. "$PRE"; [ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble at $PRE is stale or truncated: rm it and re-run this block"; exit 1; }
|
||||
```
|
||||
|
||||
Every later Bash call that touches the API starts with the same two loader lines from
|
||||
@@ -379,7 +426,7 @@ and no per-call body to hand-build.
|
||||
|
||||
```bash
|
||||
. "${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh" 2>/dev/null # §0 loader
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
N=(alpha beta) # INVENT one fresh case name per worker; never list cases first
|
||||
# (a name may carry a mode: `beta:deepseek`, see below)
|
||||
T=('reply with one line: the absolute path of your working directory'
|
||||
@@ -440,9 +487,11 @@ Four things this block leans on, each one link away, no detour needed to run it:
|
||||
skill: §5.1. Those workspaces do get hooks now, unless the operator disabled it.
|
||||
- `sendwait` supplies the `\r`, picks a fresh `seq`, and self-heals a stranded Enter.
|
||||
A prompt without the `\r` is never submitted (§3), a reused `seq` is silently
|
||||
swallowed as an already-applied duplicate, and an Enter eaten by an Ink repaint
|
||||
strands the prompt on the composer until a bare `\r` follows: all three are reasons
|
||||
to let `sendwait` build the call rather than hand-rolling it.
|
||||
swallowed as an already-applied duplicate, and a lost Enter strands the prompt on the
|
||||
composer until a bare `\r` follows: Claude Code 2.1.277 and later ignore Enter for the
|
||||
first 30 to 50 seconds after the composer paints while still taking the text, so
|
||||
`sendwait` reads the composer and keeps pressing Enter until the prompt has left it.
|
||||
All three are reasons to let `sendwait` build the call rather than hand-rolling it.
|
||||
- Each `sendwait` costs that worker one billed turn, as does every prompt you send it.
|
||||
- Deleting the sessions does **not** remove the case directories. They are marked as
|
||||
agent-created, so `GET /api/v1/cases/agent-created` lists them for cleanup: §5.14.
|
||||
|
||||
+66
-19
@@ -1,4 +1,4 @@
|
||||
# ---- Codeman agent preamble 1.22.0 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
# ---- Codeman agent preamble 1.30.1 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
API="${CODEMAN_API_URL:?CODEMAN_API_URL not set; refusing to guess}"
|
||||
SELF="${CODEMAN_SESSION_ID:?CODEMAN_SESSION_ID not set}"
|
||||
# Credentials, cheapest first. Your session has usually INHERITED the server's
|
||||
@@ -72,6 +72,27 @@ _trust_key() { # <sid> -> "confirm" | "move" | "" (nothing safe to press)
|
||||
| tr -d ' \t' | grep -i '❯[0-9.]*\(yes,itrustthisfolder\|no,exit\)' | tail -1 \
|
||||
| sed -e 's/.*[Yy]es,.*/confirm/' -e 's/.*[Nn]o,.*/move/'
|
||||
}
|
||||
# ---- the composer: is the prompt still sitting there, unsent? ----
|
||||
# ⚠️ Claude Code 2.1.277 (auto-installed 2026-09-18) takes typed text the moment the
|
||||
# composer paints but IGNORES Enter for the first 30-50 seconds after it: the \r that
|
||||
# Codeman sends 50 ms after the text and a lone nudge at 20 s both leave the prompt
|
||||
# stranded, with `0 tokens`, while the wait burns its whole timeout. Measured through
|
||||
# this very route: Enter at 28 s stranded, Enter at 51 s submitted. So sendwait READS
|
||||
# the composer and keeps pressing Enter while the prompt is still there.
|
||||
_composer_text() { # <sid> -> the composer's text with ALL whitespace removed: "" once
|
||||
# the prompt was taken, "?" when the pane shows no composer at all. The composer is
|
||||
# the LAST `❯` line: Claude Code echoes a submitted prompt with the same glyph higher
|
||||
# up in the transcript, so only the last one says whether the text was taken.
|
||||
local t
|
||||
t=$("${CURL[@]}" -G "$API/api/v1/sessions/$1/terminal" --data-urlencode 'full=1' \
|
||||
| jq -r '.data.terminalBuffer // empty' \
|
||||
| sed -e "s/$(printf '\033')\[[0-9;?]*[a-zA-Z]//g" -e "s/$(printf '\033')[()][AB0]//g" \
|
||||
| tr -d '\r' | grep -a '^[[:space:]]*❯' | tail -1)
|
||||
[ -n "$t" ] || { printf '?'; return 0; }
|
||||
# Claude Code draws a NO-BREAK SPACE (U+00A0) after the glyph, which [:space:] does
|
||||
# not cover, so it is stripped by its bytes, portably (BSD sed has no \xHH).
|
||||
printf '%s' "$t" | sed 's/^[[:space:]]*❯//' | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g"
|
||||
}
|
||||
_accept_trust() { # <sid> -> 0 once it has answered the dialog, 1 if it could not
|
||||
local sid="$1" k i=1
|
||||
while [ "$i" -le 6 ]; do
|
||||
@@ -184,12 +205,16 @@ spawn_workers() {
|
||||
# worker a silent no-op that still "succeeds" and reports the previous turn's state.
|
||||
# Pass seq explicitly for exactly one reason: resending a possibly-delivered frame as a
|
||||
# deliberate duplicate, at the SAME number (§5.3).
|
||||
# Delivery is SELF-HEALING: an Ink repaint occasionally eats the Enter, leaving the
|
||||
# typed prompt stranded on the composer while a long wait runs its whole timeout
|
||||
# (observed live). So the first wait is short; on its timeout a bare \r goes out (the
|
||||
# missing Enter when the prompt is stranded, a no-op when the turn is genuinely
|
||||
# running), then the ORIGINAL frame is resent unchanged, which the server takes as a
|
||||
# tagged duplicate: it re-waits without retyping (§5.3). Trustworthy for a worker
|
||||
# Delivery is SELF-HEALING: the Enter can be lost (an Ink repaint eats it, and Claude
|
||||
# Code 2.1.277+ ignores it outright for the first 30-50 s after the composer paints),
|
||||
# leaving the typed prompt stranded on the composer while a long wait runs its whole
|
||||
# timeout (observed live, twelve reviews in a row). So the first wait is short; on its
|
||||
# timeout the ORIGINAL frame is resent unchanged as a long re-wait (a tagged duplicate:
|
||||
# the server re-waits without retyping, §5.3) and kept open in the background, while
|
||||
# the composer is READ (_composer_text) and, as long as the prompt is still sitting
|
||||
# there, a bare \r goes out about every ten seconds, up to twelve times. An empty
|
||||
# composer ends the loop, so a prompt that was taken is never nudged again, and the
|
||||
# wait that was open the whole time is what reports the turn's end. Trustworthy for a worker
|
||||
# spawn_worker handed back -- claude (hooks vetted) or deepseek (status bridge) --
|
||||
# and for those only. Hook-less workspaces and the other modes resolve on flapping
|
||||
# idle: markers instead (§5.5). ⚠️ A dsh worker running a profile that does not
|
||||
@@ -197,7 +222,7 @@ spawn_workers() {
|
||||
# it accepts the send and then burns both waits. One timeout on a dsh worker whose
|
||||
# pane clearly finished means that profile, so switch that worker to markers.
|
||||
sendwait() {
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r c head n=0 tmp bg i
|
||||
# `wait:"stop,exit"`, never the `wait:true` default set: that set also carries
|
||||
# `idle`, which is INFERRED from output stabilization and flaps mid-turn. On a
|
||||
# dsh worker whose TUI repaints rarely the session reads `idle` while the model
|
||||
@@ -212,16 +237,38 @@ sendwait() {
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$body")
|
||||
if jq -e '.data.delivered and .data.wait.timedOut' <<<"$r" >/dev/null 2>&1; then
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
# The resend is a tagged DUPLICATE, so the server skips the write and reports
|
||||
# `delivered:false` for it -- truthfully, but about the wrong send. The first
|
||||
# one delivered, so carry that forward, or §1's cleanup reads a completed turn
|
||||
# as an undelivered one and keeps a finished worker forever.
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" \
|
||||
| jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end')
|
||||
# ⚠️ The long re-wait is registered FIRST and stays open for the rest of this call,
|
||||
# in the background, while the Enter loop below works the composer. Signals have
|
||||
# no history: a `stop` that fires while no wait is open (during a composer read
|
||||
# between two short waits, measured) is lost, and the next wait then runs its
|
||||
# whole timeout on a turn that already ended. The resend is a tagged DUPLICATE,
|
||||
# so the server skips the write and re-waits without retyping (§5.3).
|
||||
tmp=$(mktemp "${TMPDIR:-/tmp}/codeman-wait.XXXXXX") || return 1
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" > "$tmp" &
|
||||
bg=$!
|
||||
# The prompt's head with whitespace removed, matched literally (the "$head"
|
||||
# quoting inside ${c#...} keeps a * or ? in the prompt from acting as a glob).
|
||||
head=$(printf '%s' "$p" | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g" | head -c 24)
|
||||
while [ "$n" -lt 12 ] && [ ! -s "$tmp" ]; do # a non-empty file means the wait ended
|
||||
c=$(_composer_text "$sid")
|
||||
if [ "$c" = '?' ]; then
|
||||
[ "$n" -eq 0 ] || break # unreadable pane: one Enter, then trust it
|
||||
elif [ -z "$head" ] || [ "${c#"$head"}" = "$c" ]; then
|
||||
break # composer empty (taken) or holding other text
|
||||
fi
|
||||
n=$((n+1))
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
i=0; while [ "$i" -lt 10 ] && [ ! -s "$tmp" ]; do sleep 1; i=$((i+1)); done
|
||||
done
|
||||
wait "$bg"
|
||||
# The duplicate reports `delivered:false` -- truthfully, but about the wrong send.
|
||||
# The first one delivered, so carry that forward, or §1's cleanup reads a completed
|
||||
# turn as an undelivered one and keeps a finished worker forever.
|
||||
r=$(jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end' < "$tmp")
|
||||
rm -f "$tmp"
|
||||
fi
|
||||
printf '%s\n' "$r"
|
||||
}
|
||||
@@ -247,4 +294,4 @@ last_text() {
|
||||
# The stamp is the LAST line on purpose (a truncated write leaves it unset) and is kept
|
||||
# bare on purpose: the write condition above anchors on it with $, so an inline comment
|
||||
# here would fail that match and rewrite this file on every single bootstrap.
|
||||
CODEMAN_PREAMBLE=1.22.0
|
||||
CODEMAN_PREAMBLE=1.30.1
|
||||
|
||||
@@ -340,7 +340,7 @@ ESC=$(printf '\033')
|
||||
### Starting a worker
|
||||
|
||||
`POST /api/v1/quick-start` body (all optional):
|
||||
`{"caseName":"worker-1","mode":"claude","sessionName":"w9-worker","effort":"high"}`
|
||||
`{"caseName":"worker-1","mode":"claude","sessionName":"auth-worker","effort":"high"}`
|
||||
, `mode` ∈ `claude|shell|opencode|codex|gemini|antigravity|pi|grok|deepseek|omp`; response is
|
||||
`.data.{sessionId, caseName, casePath}`. Creates the case directory (a real directory
|
||||
on the user's disk) if missing, do not retry it in a loop, and remember the name.
|
||||
|
||||
@@ -101,11 +101,16 @@ the case name, read it from the listing.
|
||||
From Codeman 1.16 a LOCAL claude spawn passes `--name <session name>` when the local
|
||||
CLI is 2.1.224+ (`buildNameCliArgs`, `session-cli-builder.ts:97-101`, wired in at
|
||||
`tmux-manager.ts:797`), so a worker's peer name usually IS its Codeman session name
|
||||
(verified live: quick-start with `sessionName: "w9-msgtest"` listed as `w9-msgtest`,
|
||||
and its messages arrive tagged `from-name="w9-msgtest"`; a derived-name worker's
|
||||
(verified live: a quick-start `sessionName` is listed as that exact peer name, and
|
||||
the worker's messages arrive tagged `from-name="<that name>"`; a derived-name worker's
|
||||
messages carry no `from-name`). Name your workers: a quick-start WITHOUT
|
||||
`sessionName` leaves the Codeman name empty, so there is nothing to pass and the
|
||||
peer name stays derived. The flag is fail-closed (older/unknown CLI omits it, because an
|
||||
peer name stays derived. ⚠️ Give them a DESCRIPTIVE name: only a name the user chose
|
||||
is pinned (`Session.cliPinnedName`), because `--name` is also the conversation's
|
||||
`/resume` title and terminal title and suppresses Claude's own generated title. A
|
||||
placeholder-shaped name (`w9-msgtest`, anything matching `isGeneratedSessionName`)
|
||||
and an auto name are NOT passed, so such a worker's peer name is derived; use
|
||||
`msgtest-worker` rather than `w9-msgtest`. The flag is fail-closed (older/unknown CLI omits it, because an
|
||||
unknown flag aborts startup and would kill every spawn) and allowlist-sanitized (a name of
|
||||
only unsafe characters is dropped), and the docker/remote builders never see it at all
|
||||
(`tmux-manager.ts:782-789`), which is why the `tmux` column stays the canonical join key
|
||||
@@ -196,8 +201,8 @@ idle:
|
||||
The contract an orchestrator follows for any fleet of two or more messaging workers.
|
||||
Every topology in the next section is this protocol plus a wiring diagram.
|
||||
|
||||
1. **Spawn with a name, and confirm hooks.** Use `quick-start` with `sessionName` (the
|
||||
`--name` gate above). Session create installs the hooks block into the workspace
|
||||
1. **Spawn with a name, and confirm hooks.** Use `quick-start` with a descriptive,
|
||||
non-`w<N>-` `sessionName` (the `--name` gate above). Session create installs the hooks block into the workspace
|
||||
whatever kind it is, so a linked case and a raw `POST /api/sessions` path both get
|
||||
`stop`/`blocked` by default. ⚠️ Not unconditionally: the operator can turn
|
||||
`workspaceHooksEnabled` off, remote SSH sessions never get hooks, and a session from
|
||||
|
||||
@@ -21,7 +21,7 @@ by sourcing the preamble file the §0 bootstrap wrote, and checking its version
|
||||
|
||||
```bash
|
||||
. "${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh" 2>/dev/null
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble missing or stale; re-run the §0 bootstrap"; exit 1; }
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble missing or stale; re-run the §0 bootstrap"; exit 1; }
|
||||
```
|
||||
|
||||
Do **not** re-paste the preamble body into each call. Sourcing it is what retires the
|
||||
|
||||
@@ -692,8 +692,9 @@ The shape, each step verified live (probes, failure modes and safety detail in
|
||||
[§5.2](#52-readiness)).
|
||||
2. `ListAgents`: find the worker's row by its `tmux codeman-<first 8 of session id>`
|
||||
column; the row's `name [ref]` is the address. On Codeman 1.16+ with claude
|
||||
2.1.224+ a worker's peer name is its Codeman session name, so pass `sessionName`
|
||||
in quick-start to pick it; older setups list a name derived from the case folder.
|
||||
2.1.224+ a worker's peer name is its Codeman session name, so pass a DESCRIPTIVE
|
||||
`sessionName` in quick-start to pick it (a `w<N>-` placeholder-shaped name is not
|
||||
pinned, so it lists derived); older setups list a name derived from the case folder.
|
||||
No row = messaging is off for that worker (it is feature-flagged even on matching
|
||||
CLI versions, observed live): fall back to the HTTP recipes without complaint.
|
||||
3. `SendMessage` the task; first contact must use the `name [ref]` form copied from
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
/**
|
||||
* @fileoverview Carry a Codeman rename into Claude Code's own session title.
|
||||
*
|
||||
* Claude Code keeps a conversation's title in its transcript as a
|
||||
* `{"type":"custom-title"}` row (what `/rename` writes), last row wins, and the
|
||||
* `/resume` picker shows `customTitle ?? aiTitle`. Renaming a tab in Codeman
|
||||
* used to change only the tab, so `/resume` kept listing the old name.
|
||||
*
|
||||
* Appending the row is enough for a pane that was spawned WITHOUT `--name`
|
||||
* (every placeholder- or auto-named tab, see `Session.cliPinnedName`): that
|
||||
* process holds no title of its own and never writes one back. A process that
|
||||
* WAS spawned with `--name` re-appends its in-memory title after each turn, so
|
||||
* there the new title holds from the next spawn, which pins the new name.
|
||||
*
|
||||
* @module claude-session-title
|
||||
*/
|
||||
|
||||
import fs from 'node:fs/promises';
|
||||
|
||||
/**
|
||||
* Append a `custom-title` row for `conversationId` to an existing transcript.
|
||||
* Never creates the file: a missing transcript means the conversation has not
|
||||
* been written yet, and a file of only a title row would show up in `/resume`
|
||||
* as an empty conversation. Returns whether a row was written.
|
||||
*/
|
||||
export async function appendClaudeCustomTitle(
|
||||
transcriptPath: string,
|
||||
conversationId: string,
|
||||
title: string
|
||||
): Promise<boolean> {
|
||||
const customTitle = title.trim();
|
||||
// Claude reads the row through `customTitle ?? aiTitle`, so an empty string
|
||||
// would blank the picker entry rather than fall back to the generated title.
|
||||
if (!customTitle) return false;
|
||||
try {
|
||||
if (!(await fs.stat(transcriptPath)).isFile()) return false;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
// One O_APPEND write of one line, the same way Claude appends its own rows,
|
||||
// so it cannot interleave with a row the live process is writing.
|
||||
const row = JSON.stringify({ type: 'custom-title', customTitle, sessionId: conversationId });
|
||||
await fs.appendFile(transcriptPath, `${row}\n`);
|
||||
return true;
|
||||
}
|
||||
+10
@@ -129,6 +129,8 @@ program
|
||||
|
||||
/** Same registry the server resolves case names through (mirrors `case-routes.ts`). */
|
||||
const LINKED_CASES_FILE = dataPath('linked-cases.json');
|
||||
/** Graceful shutdown budget before the process force-exits (see the SIGTERM handler). */
|
||||
const SHUTDOWN_FORCE_EXIT_MS = 10_000;
|
||||
|
||||
/**
|
||||
* Case name to directory, checking `linked-cases.json` FIRST and falling back to the
|
||||
@@ -1002,6 +1004,14 @@ webCmd.action(async (options) => {
|
||||
if (shuttingDown) return;
|
||||
shuttingDown = true;
|
||||
console.log(palette.warn(`\n${signal} received, shutting down gracefully...`));
|
||||
// A hung stop() must not keep the process alive: the listener is already
|
||||
// closed by then, and a KeepAlive LaunchDaemon only respawns the server once
|
||||
// it EXITS (systemd would SIGKILL after TimeoutStopSec; launchd does not).
|
||||
// Seen after a self-update on macOS: port closed, process alive, service down.
|
||||
setTimeout(() => {
|
||||
console.error(palette.err(`Shutdown did not finish in ${SHUTDOWN_FORCE_EXIT_MS / 1000}s, forcing exit`));
|
||||
process.exit(1);
|
||||
}, SHUTDOWN_FORCE_EXIT_MS).unref();
|
||||
try {
|
||||
await server.stop();
|
||||
} catch (err) {
|
||||
|
||||
@@ -0,0 +1,115 @@
|
||||
/**
|
||||
* @fileoverview Write side of the CLI registry (docs/cli-enable-disable-plan.md, Phases 3/5).
|
||||
*
|
||||
* Kept deliberately SEPARATE from `registry.ts`, whose reading path does no writes on import
|
||||
* (`schemas.ts` imports it, transitively). Only `cli-registry-routes.ts` imports this module,
|
||||
* so that property still holds for every OTHER importer of the registry.
|
||||
*
|
||||
* Every mutation goes through `mutateRegistryFile()`, which does three things the #476 review
|
||||
* found missing:
|
||||
*
|
||||
* - **Serialized.** Mutations run one at a time on a single promise chain, and each one
|
||||
* reads, changes, writes and reloads before the next starts. Unserialized read-modify-write
|
||||
* lost toggles when three `PUT /api/clis/:id` calls ran in parallel.
|
||||
* - **Refuses a file it must not trust.** The reader ignores a `clis.json` with any
|
||||
* group/world permission bit and quarantines one that does not parse. The writer used to
|
||||
* treat both as "start fresh", so one Settings click replaced a hand-edited file with a
|
||||
* one-key file, or rewrote a refused file as 0600 and so trusted it. It now starts fresh
|
||||
* ONLY on ENOENT and otherwise throws `RegistryWriteRefusedError`, leaving the file alone.
|
||||
* - **Unique temp file.** Every write gets its own tmp name before the rename, so two writes
|
||||
* can never rename each other's temp file away (the ENOENT-on-rename 500s).
|
||||
*
|
||||
* Same tmp+rename+0600 shape as `custom-model-hosts.ts`. The file is hand-editable, so a
|
||||
* write must never leave it half-written, and 0600 is the mode `isUnsafePermissions()`
|
||||
* requires on the next read.
|
||||
*/
|
||||
|
||||
import { randomUUID } from 'node:crypto';
|
||||
import { existsSync, mkdirSync } from 'node:fs';
|
||||
import fs from 'node:fs/promises';
|
||||
import { dirname } from 'node:path';
|
||||
import { isUnsafePermissions, registryFilePath, reloadCliRegistry } from './registry.js';
|
||||
import type { CliRegistryFile } from './types.js';
|
||||
|
||||
/** A write refused because the existing `clis.json` must not be overwritten. The message is user-facing. */
|
||||
export class RegistryWriteRefusedError extends Error {
|
||||
constructor(message: string) {
|
||||
super(message);
|
||||
this.name = 'RegistryWriteRefusedError';
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Read the raw override file for mutation. Only a MISSING file starts fresh. A file with
|
||||
* unsafe permissions, one that cannot be read, or one that does not parse is refused rather
|
||||
* than overwritten, because the user's hand-edit is worth more than one toggle.
|
||||
*/
|
||||
export async function readRegistryFileForWrite(): Promise<CliRegistryFile> {
|
||||
const path = registryFilePath();
|
||||
let raw: string;
|
||||
try {
|
||||
raw = await fs.readFile(path, 'utf-8');
|
||||
} catch (err) {
|
||||
if ((err as NodeJS.ErrnoException).code === 'ENOENT') return { schemaVersion: 1, clis: {} };
|
||||
throw new RegistryWriteRefusedError(`Cannot read ${path} (${(err as Error).message}); not changing it.`);
|
||||
}
|
||||
if (isUnsafePermissions(path)) {
|
||||
throw new RegistryWriteRefusedError(
|
||||
`${path} has group/world permission bits, so Codeman ignores it. Run \`chmod 600 ${path}\` and check its contents before changing CLIs here.`
|
||||
);
|
||||
}
|
||||
let parsed: unknown;
|
||||
try {
|
||||
parsed = JSON.parse(raw);
|
||||
} catch (err) {
|
||||
throw new RegistryWriteRefusedError(
|
||||
`${path} is not valid JSON (${(err as Error).message}). Fix or remove it before changing CLIs here.`
|
||||
);
|
||||
}
|
||||
const clis = (parsed as { clis?: unknown } | null)?.clis;
|
||||
if (typeof parsed !== 'object' || parsed === null || typeof clis !== 'object' || clis === null) {
|
||||
throw new RegistryWriteRefusedError(`${path} has no "clis" object. Fix or remove it before changing CLIs here.`);
|
||||
}
|
||||
return parsed as CliRegistryFile;
|
||||
}
|
||||
|
||||
export async function writeRegistryFile(file: CliRegistryFile): Promise<void> {
|
||||
const target = registryFilePath();
|
||||
const dir = dirname(target);
|
||||
if (!existsSync(dir)) mkdirSync(dir, { recursive: true });
|
||||
const tmp = `${target}.${process.pid}.${randomUUID()}.tmp`;
|
||||
try {
|
||||
await fs.writeFile(tmp, JSON.stringify(file, null, 2), { mode: 0o600 });
|
||||
await fs.rename(tmp, target);
|
||||
} catch (err) {
|
||||
await fs.rm(tmp, { force: true }).catch(() => {});
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
let mutationChain: Promise<unknown> = Promise.resolve();
|
||||
|
||||
/**
|
||||
* Run one registry mutation. The chain holds exactly one at a time: `fn` receives the
|
||||
* current file and returns `{ file, result }`. If `file` is set it is written and the
|
||||
* registry reloaded before the next mutation starts; if not, nothing is written, which is
|
||||
* how a validation failure returns early. Checks made inside `fn` (does this id exist,
|
||||
* is it a duplicate) therefore see every earlier mutation's result.
|
||||
*
|
||||
* A failed mutation rejects its own caller only. The chain keeps going.
|
||||
*/
|
||||
export function mutateRegistryFile<T>(
|
||||
fn: (file: CliRegistryFile) => Promise<{ file?: CliRegistryFile; result: T }> | { file?: CliRegistryFile; result: T }
|
||||
): Promise<T> {
|
||||
const run = mutationChain.then(async () => {
|
||||
const current = await readRegistryFileForWrite();
|
||||
const { file, result } = await fn(current);
|
||||
if (file) {
|
||||
await writeRegistryFile(file);
|
||||
reloadCliRegistry();
|
||||
}
|
||||
return result;
|
||||
});
|
||||
mutationChain = run.catch(() => {});
|
||||
return run;
|
||||
}
|
||||
@@ -48,6 +48,16 @@ function filePath(): string {
|
||||
return dataPath('clis.json');
|
||||
}
|
||||
|
||||
/**
|
||||
* The resolved path of `~/.codeman/clis.json`, exported for the write API
|
||||
* (`cli-registry-writer.ts`, docs/cli-enable-disable-plan.md Phases 3/5) so both the read and
|
||||
* write sides resolve the SAME path through the SAME instance-scoped helper — never a second
|
||||
* `dataPath('clis.json')` call that could drift from this one under a future `dataPath()` change.
|
||||
*/
|
||||
export function registryFilePath(): string {
|
||||
return filePath();
|
||||
}
|
||||
|
||||
/**
|
||||
* Keys that must never be merged out of a hand-editable JSON file.
|
||||
*
|
||||
@@ -90,8 +100,11 @@ export interface LoadResult {
|
||||
* as mode 0o666 there regardless of its actual ACL), so this check would flag every file on
|
||||
* Windows and silently ignore all user config. `win32` relies on NTFS ACLs instead, which
|
||||
* this check cannot see and does not attempt to.
|
||||
*
|
||||
* Exported for `registry-writer.ts`, which must refuse the same files: rewriting a refused
|
||||
* file as 0600 would silently turn it into trusted config.
|
||||
*/
|
||||
function isUnsafePermissions(path: string): boolean {
|
||||
export function isUnsafePermissions(path: string): boolean {
|
||||
if (process.platform === 'win32') return false;
|
||||
try {
|
||||
const mode = statSync(path).mode & 0o777;
|
||||
|
||||
@@ -294,6 +294,23 @@ const capabilitiesSchema = z
|
||||
effort: z.boolean(),
|
||||
agentSkillInjection: z.boolean(),
|
||||
statusLineTelemetry: z.boolean(),
|
||||
// How many columns this CLI indents its transcript body by, so a copy can take
|
||||
// that much off the clipboard. Bounded, because it is the whole strip: a copy
|
||||
// never removes more than this, nor more than every selected line shares.
|
||||
//
|
||||
// ⚠ DECLARED, not measured off the pane, and two measured attempts are why.
|
||||
// Asking whether the pane painted spaces across the unused part of each row
|
||||
// separates a TUI from a shell perfectly where it fires and never
|
||||
// over-stripped, but it is a function of pane WIDTH: that padding exists
|
||||
// only while a rendered line stops short of the CLI's own layout width, and
|
||||
// Claude Code's prose wraps to fill it — the share of padded rows on one
|
||||
// live transcript ran 44%, 6%, 6%, 7% and 87% at 123, 160, 198, 235 and 298
|
||||
// columns, so the strip did nothing at any ordinary size. Taking the
|
||||
// narrowest indent on screen instead fires everywhere and over-strips, since
|
||||
// a file listing inside the transcript can be the narrowest thing on it.
|
||||
// A declared width cannot do either. Absent means no strip, so a CLI whose
|
||||
// transcript layout nobody has measured is never touched.
|
||||
transcriptGutter: z.number().int().min(1).max(8).optional(),
|
||||
workDetect: z
|
||||
.object({
|
||||
promptGlyph: z.string().min(1).max(8),
|
||||
@@ -308,8 +325,38 @@ const capabilitiesSchema = z
|
||||
(src) => compileVersionRegex(src) !== null,
|
||||
'workingLine must be a regex compileVersionRegex() accepts: at most 200 characters, no nested quantifiers'
|
||||
),
|
||||
// Same guard, same reasons: this one runs over the foot of a pane capture every
|
||||
// time a session settles, and ~/.codeman/clis.json can set it.
|
||||
watchingLine: z
|
||||
.string()
|
||||
.min(1)
|
||||
.refine(
|
||||
(src) => compileVersionRegex(src) !== null,
|
||||
'watchingLine must be a regex compileVersionRegex() accepts: at most 200 characters, no nested quantifiers'
|
||||
)
|
||||
.optional(),
|
||||
// Bounded hard: this is how far up the screen a config file may push the search,
|
||||
// and every row it adds is one more row the agent itself may be able to write.
|
||||
watchingLines: z.number().int().min(1).max(8).optional(),
|
||||
// Same guard again: tested against a pane row every time a session settles.
|
||||
awaitingLine: z
|
||||
.string()
|
||||
.min(1)
|
||||
.refine(
|
||||
(src) => compileVersionRegex(src) !== null,
|
||||
'awaitingLine must be a regex compileVersionRegex() accepts: at most 200 characters, no nested quantifiers'
|
||||
)
|
||||
.optional(),
|
||||
})
|
||||
.strict()
|
||||
// A window with nothing to search is a typo, not a configuration. Refused at LOAD
|
||||
// time for the same reason `privilegedParams[].param` is checked against the params
|
||||
// the entry declares: the failure is otherwise silent and looks like a feature that
|
||||
// simply never fires.
|
||||
.refine(
|
||||
(v) => v.watchingLines === undefined || v.watchingLine !== undefined,
|
||||
'watchingLines has nothing to bound without a watchingLine'
|
||||
)
|
||||
.optional(),
|
||||
model: z
|
||||
.object({ source: z.enum(['flag', 'claude-settings-file', 'none']), param: z.string().optional() })
|
||||
@@ -340,6 +387,35 @@ const capabilitiesSchema = z
|
||||
// an env var, so it declares baseUrl/apiKey injection with no model var at all.
|
||||
modelVars: z.array(envName).max(8),
|
||||
launchModel: launchModelTemplate,
|
||||
// Optional: the env var to carry a discovered per-model context-window size
|
||||
// (claude's CLAUDE_CODE_MAX_CONTEXT_TOKENS), and/or the env var that isolates
|
||||
// this session's config/credential directory from the user's real one (claude's
|
||||
// CLAUDE_CONFIG_DIR) so an injected API key never collides with a stored OAuth
|
||||
// session. See the customModelInjection doc comment in cli-registry/types.ts.
|
||||
contextLengthVar: envName.optional(),
|
||||
configDirVar: envName.optional(),
|
||||
// Relative path, WITHIN the isolated configDirVar directory, of a trust-dialog
|
||||
// seed file the CLI itself owns the shape of — claude's `.claude.json`
|
||||
// `customApiKeyResponses.approved` list, the same field an interactive "Detected
|
||||
// a custom API key — use it?" prompt writes to on a real terminal. Only makes
|
||||
// sense alongside configDirVar (an isolated, otherwise-empty directory has none
|
||||
// of a real profile's prior approvals), and only implemented for the
|
||||
// 'claude-api-key-responses' shape today — see custom-model-injection-apply.ts.
|
||||
apiKeyTrustFile: z
|
||||
.object({ relPath: z.string().min(1).max(80), shape: z.literal('claude-api-key-responses') })
|
||||
.strict()
|
||||
.optional(),
|
||||
// An isolated config directory replays the CLI's whole first-run sequence (theme
|
||||
// picker, security notes, per-project trust dialog, bypass-permissions warning)
|
||||
// on every launch, same root cause as apiKeyTrustFile above — this reuses that
|
||||
// same file to pre-seed the state a real, already-onboarded profile carries. See
|
||||
// the customModelInjection doc comment in cli-registry/types.ts.
|
||||
skipFirstRunPrompts: z.boolean().optional(),
|
||||
// DeepSeek-only, confirmed by reading its own bundled SDK source: it concatenates
|
||||
// "/chat/completions" onto baseUrlVar's value with no "/v1" of its own, while
|
||||
// llama-swap/llama.cpp only serves the "/v1/..." path — claude/gemini must NOT
|
||||
// get this. See the customModelInjection doc comment in cli-registry/types.ts.
|
||||
appendV1Suffix: z.boolean().optional(),
|
||||
})
|
||||
.strict(),
|
||||
z
|
||||
|
||||
@@ -75,11 +75,24 @@ function agentDefaults(): Pick<
|
||||
};
|
||||
}
|
||||
|
||||
// `accent` on every entry below (except SHELL, which the frontend renders no
|
||||
// distinct color for) is measured from the actual `.btn-toolbar.btn-run.mode-<id>`
|
||||
// CSS rule's `border-color` on the OG skin (styles.css) — the single cleanest
|
||||
// representative hex each entry's own multi-stop gradient resolves around.
|
||||
// Corrected 2026-09-21 after PR #458's review found several were simply wrong
|
||||
// (e.g. claude was registered as Anthropic's brand orange, `#d97757`, but the
|
||||
// button renders blue): `docs/cli-registry.md`'s own "transcribed, not
|
||||
// authoritative, re-measure before wiring one up" warning for this
|
||||
// DECLARED-FOR-LATER field, taken literally. The one exception is GEMINI, whose
|
||||
// run-button border (#60a5fa) is the only one that disagrees with its own tab badge
|
||||
// and run-mode dot (#8ab4f8); it takes the badge colour, so every accent names the
|
||||
// same hex the frontend uses as that CLI's flat identity. This is a data-accuracy fix only —
|
||||
// `accent` still has no reader, so nothing rendered changes because of it.
|
||||
const CLAUDE: CliEntry = {
|
||||
id: 'claude' as CliEntry['id'],
|
||||
label: 'Claude',
|
||||
label: 'Claude Code',
|
||||
shortBadge: 'CC',
|
||||
accent: '#d97757',
|
||||
accent: '#3b82f6',
|
||||
enabled: true,
|
||||
stock: true,
|
||||
order: 0,
|
||||
@@ -200,12 +213,48 @@ const CLAUDE: CliEntry = {
|
||||
},
|
||||
capabilities: {
|
||||
external: false,
|
||||
// Claude indents its transcript body two columns and puts its own ●/✻/❯ markers
|
||||
// in them, so a copy can drop two and paste flush. Claude and codex are the only
|
||||
// entries that declare this, because theirs are the only gutters that have been measured.
|
||||
transcriptGutter: 2,
|
||||
// The historical hard-coded pair, now stated as data. `workingLine` matches both the
|
||||
// `✻ Actualizing… (39s · ↓ 2.0k tokens)` status line and the bare `esc to interrupt`
|
||||
// footer, because tmux repaints partially and only one of the two may land in a chunk.
|
||||
workDetect: {
|
||||
promptGlyph: '❯',
|
||||
workingLine: String.raw`…\s*\((?:\d+h\s+)?(?:\d+m\s+)?\d+s\b|esc to interrupt`,
|
||||
// Claude prints what it started in the background on the footer row beneath its
|
||||
// composer, as `⏵⏵ bypass permissions on · 1 monitor · ← for agents`. The labels are
|
||||
// the CLI's own words for each kind of background task, and group 1 is the one
|
||||
// Codeman badges the session with. Verified against a live 2.1.278 pane on
|
||||
// 2026-09-21.
|
||||
// ⚠️ Two things keep an agent from writing its own label here, and both matter.
|
||||
// The footer is the LAST row, so the default one-row window (`WATCHING_TAIL_LINES`)
|
||||
// holds nothing but Ink's own chrome — in particular it leaves out the status line
|
||||
// directly above, whose content comes from a `statusLine` command a bypassed
|
||||
// session can write into its own `.claude/settings.json`. And the leading `·` keeps
|
||||
// the match on the footer's own item list rather than on any text that happens to
|
||||
// carry a count. A footer that ever drew the chip as its only item would report no
|
||||
// watching rather than open that door. See `watchingLabel()` in
|
||||
// `session-activity.ts`.
|
||||
// ⚠️ An Artifact comment monitor is the one chip that waits on the user. The agent
|
||||
// has published a page and hears nothing until somebody comments on it, so the
|
||||
// lookahead refuses the whole row while that chip is on it, whatever else is
|
||||
// running beside it. The `^` is what makes the lookahead judge the row once:
|
||||
// without it the engine retries from each later position, and a start past the
|
||||
// chip reports the shell beside it. The lookahead keys on "Artifact" alone, so a
|
||||
// footer cut off mid-chip (`· 1 Artifact…`, `· 1 Artifact comm…`) is still refused;
|
||||
// no other chip on this row says "Artifact". Counting the chip as watching kept the
|
||||
// idle alert quiet for a session that was waiting for a human.
|
||||
watchingLine: String.raw`^(?!.*Artifact).*?·\s*(\d+ (?:monitors?|shells?|teams?|local agents?|cloud sessions?|MCP tasks?|background tasks?|(?:background|remote) dynamic workflows?))`,
|
||||
// When a turn ends while background agents or an ultracode workflow are still
|
||||
// running, Claude swaps its `✻ Brewed for 1m 18s` closing row for
|
||||
// `✻ Waiting for 2 background agents and 1 dynamic workflow to finish` and resumes
|
||||
// by itself when they report back. Read from the 2.1.283 bundle (the turn-duration
|
||||
// renderer) and a live pane on 2026-09-28. The row is a snapshot taken at turn end
|
||||
// and never redrawn, which is why only the newest row above the composer counts.
|
||||
// Anchored on column 0: Claude's own rows start there, the agent's prose never does.
|
||||
awaitingLine: String.raw`^✻ Waiting for \d+ (?:background agents?|dynamic workflows?)\b`,
|
||||
},
|
||||
requiresMux: false,
|
||||
// Claude installs Codeman's own hooks block into every workspace it runs in, so its
|
||||
@@ -214,6 +263,8 @@ const CLAUDE: CliEntry = {
|
||||
transcript: 'claude-jsonl',
|
||||
altScreen: 'strip-full',
|
||||
echo: { policy: 'buffer', anchor: { kind: 'glyph', glyph: '❯', offset: 2 } },
|
||||
// Declared-for-later: the live rule (`_shouldForwardWheelToApp`, terminal-ui.js) is this version
|
||||
// AND the server-published `cliMouseTracking` flag (#498), so wiring this field up needs both.
|
||||
wheelForward: { mode: 'version-gated', minVersion: '2.1.187' },
|
||||
keyboardAccessory: 'agent',
|
||||
privilegedCommandGate: false,
|
||||
@@ -228,15 +279,31 @@ const CLAUDE: CliEntry = {
|
||||
privilegedParams: [],
|
||||
// ANTHROPIC_* is NOT in allowedPrefixes/allowedKeys above (deliberately — see the
|
||||
// allowedPrefixes comment nearby), so these are unreachable via plain envOverrides
|
||||
// today; listed here only so the dedicated custom-model route (docs/custom-model-endpoints-plan.md
|
||||
// chunk 5) clamps them for a non-granted multi-user owner the same way every other
|
||||
// CLI's injection vars are clamped, the day that route widens who can set them.
|
||||
// today. privilegedEnvKeys has exactly one consumer, ownerClampedEnvKeys() in
|
||||
// session-env-clamp.ts, which feeds the generic envOverrides clamp on
|
||||
// POST /api/sessions, POST /api/quick-start and reboot-restore — no custom-model
|
||||
// route reads this field at all, and the values it injects are merged in AFTER
|
||||
// that clamp runs regardless of what's listed here.
|
||||
privilegedEnvKeys: [
|
||||
'ANTHROPIC_BASE_URL',
|
||||
'ANTHROPIC_API_KEY',
|
||||
'ANTHROPIC_DEFAULT_SONNET_MODEL',
|
||||
'ANTHROPIC_DEFAULT_HAIKU_MODEL',
|
||||
'ANTHROPIC_DEFAULT_OPUS_MODEL',
|
||||
// CLAUDE_CODE_MAX_CONTEXT_TOKENS already matches the CLAUDE_CODE_* allowedPrefix, and
|
||||
// CLAUDE_CONFIG_DIR is already an allowed exact key (docs/wiki/Agent-CLIs.md), so both
|
||||
// were already reachable via plain envOverrides before this pair existed and this
|
||||
// feature does not strictly need either listed. They stay listed anyway, because
|
||||
// types.ts's rule ("every traffic-redirecting var this feature introduces MUST also
|
||||
// appear in privilegedEnvKeys") is meant to hold literally, not with an exception
|
||||
// carved out for the two vars that happen not to need it today. The real
|
||||
// consequence lands on the GENERIC envOverrides clamp above, not on this feature:
|
||||
// a non-granted multi-user owner can no longer set CLAUDE_CONFIG_DIR through
|
||||
// envOverrides at all (the per-client-account override, #255), and a PERSISTED one
|
||||
// is now stripped on reboot-restore for such an owner too — see
|
||||
// session-env-clamp.ts's own fileoverview.
|
||||
'CLAUDE_CODE_MAX_CONTEXT_TOKENS',
|
||||
'CLAUDE_CONFIG_DIR',
|
||||
],
|
||||
gates: { nameFlag: { minVersion: '2.1.224', failClosed: true } },
|
||||
// Custom Model Endpoint Profiles (docs/custom-model-endpoints-plan.md) — verified by hand against a real
|
||||
@@ -247,6 +314,32 @@ const CLAUDE: CliEntry = {
|
||||
baseUrlVar: 'ANTHROPIC_BASE_URL',
|
||||
apiKeyVar: 'ANTHROPIC_API_KEY',
|
||||
modelVars: ['ANTHROPIC_DEFAULT_SONNET_MODEL', 'ANTHROPIC_DEFAULT_HAIKU_MODEL', 'ANTHROPIC_DEFAULT_OPUS_MODEL'],
|
||||
// Verified via Claude Code's own docs: CLAUDE_CODE_MAX_CONTEXT_TOKENS overrides the
|
||||
// assumed context window and applies directly for a model name Claude Code doesn't
|
||||
// recognize as one of its own — exactly the custom-model case. Without it, Claude Code
|
||||
// assumes a large (200k) window for any unrecognized model id and never compacts,
|
||||
// eventually overflowing a much smaller real local context (see plan doc reasoning
|
||||
// above the interface for the confirmed failure).
|
||||
contextLengthVar: 'CLAUDE_CODE_MAX_CONTEXT_TOKENS',
|
||||
// Isolates this session's config/credential directory so an injected ANTHROPIC_API_KEY
|
||||
// never shares a directory with a stored claude.ai OAuth login — see the doc comment on
|
||||
// customModelInjection in cli-registry/types.ts for the traded-off side effect.
|
||||
configDirVar: 'CLAUDE_CONFIG_DIR',
|
||||
// ⚠️ Required alongside configDirVar, not optional in practice: verified live that an
|
||||
// isolated, otherwise-empty config directory makes claude stop at an interactive
|
||||
// "Detected a custom API key — use it?" prompt on EVERY launch, defaulting to "No" with
|
||||
// no one at the TTY to answer — silently refusing the very key this feature injected.
|
||||
// Pre-seeding this file's customApiKeyResponses.approved list (verified against a real
|
||||
// ~/.claude.json after answering the prompt once by hand) answers it in advance instead.
|
||||
apiKeyTrustFile: { relPath: '.claude.json', shape: 'claude-api-key-responses' },
|
||||
// ⚠️ Same isolated-directory root cause, one step further: verified live that on top
|
||||
// of the API-key prompt above, a fresh CLAUDE_CONFIG_DIR also replays claude's ENTIRE
|
||||
// first-run sequence on every launch — the theme picker, the security-notes screen,
|
||||
// the per-project "trust this folder?" dialog, and (running with
|
||||
// --dangerously-skip-permissions) a one-time bypass-permissions warning — none of
|
||||
// which a real, already-onboarded profile shows again. Pre-seeds that same
|
||||
// already-onboarded state instead of leaving a human to click through it.
|
||||
skipFirstRunPrompts: true,
|
||||
},
|
||||
},
|
||||
overlays: {
|
||||
@@ -326,7 +419,7 @@ const OPENCODE: CliEntry = {
|
||||
id: 'opencode' as CliEntry['id'],
|
||||
label: 'OpenCode',
|
||||
shortBadge: 'OC',
|
||||
accent: '#f59e0b',
|
||||
accent: '#10b981',
|
||||
enabled: true,
|
||||
stock: true,
|
||||
order: 10,
|
||||
@@ -412,7 +505,7 @@ const CODEX: CliEntry = {
|
||||
id: 'codex' as CliEntry['id'],
|
||||
label: 'Codex',
|
||||
shortBadge: 'CX',
|
||||
accent: '#6b7fd7',
|
||||
accent: '#a855f7',
|
||||
enabled: true,
|
||||
stock: true,
|
||||
order: 20,
|
||||
@@ -476,7 +569,43 @@ const CODEX: CliEntry = {
|
||||
// `Working (2m 49s • esc to interrupt)` above it while a turn runs. It animates no
|
||||
// braille spinner, and it never prints `esc to interrupt` at rest, so that phrase
|
||||
// alone separates a running turn from an idle one.
|
||||
workDetect: { promptGlyph: '›', workingLine: '[Ee]sc to interrupt' },
|
||||
// Codex pins a row of its own while a background terminal it started is still
|
||||
// running: ` 1 background terminal running · /ps to view · /stop to close`. Unlike
|
||||
// Claude's footer chip that row sits ABOVE the composer, which puts it third from the
|
||||
// bottom once the status line and the composer are counted, hence `watchingLines`.
|
||||
// Measured against a live codex-cli 0.154.0 pane on 2026-09-22: the row appears when
|
||||
// the terminal starts, follows the composer down as the conversation grows, and is
|
||||
// gone after `/stop`.
|
||||
// ⚠️ This entry CANNOT promise what Claude's does, and the difference is Codex's
|
||||
// layout rather than its pattern. The third row from the bottom is the chip only
|
||||
// while a terminal runs; with none running it is the last row of the transcript,
|
||||
// which the agent writes. Matching the complete row raises the bar — an assistant
|
||||
// message has to end with this exact line, to the character — but nothing here makes
|
||||
// forging it impossible, so do not read the Claude comment above as applying here.
|
||||
// What contains it is that codex declares `hooks: 'none'`: no hook event from a codex
|
||||
// session ever reaches `notePrompt()`, so there is no idle item to pre-acknowledge
|
||||
// and a forged label costs a wrong badge and nothing else. A CLI that gains hook
|
||||
// signals must not keep a pattern this soft.
|
||||
// ⚠️ Background TERMINALS are the only background work codex advertises on screen.
|
||||
// A sub-agent started without waiting outlives the turn just as a terminal does —
|
||||
// measured 2026-09-22, the sandboxed process was still running — and the pane shows
|
||||
// nothing at all for it: the last rows are the composer and the status line, and
|
||||
// `Sub-agents running` lives in the on-demand `/subagents` panel, not above the
|
||||
// composer. So a codex session waiting on a sub-agent reads as plainly idle here.
|
||||
// Nothing is misfiled by that (codex raises no idle prompts), and there is no row to
|
||||
// match until codex pins one.
|
||||
workDetect: {
|
||||
promptGlyph: '›',
|
||||
workingLine: '[Ee]sc to interrupt',
|
||||
watchingLine: String.raw`^\s{0,4}(\d+ background terminals?) running · /ps to view · /stop to close$`,
|
||||
watchingLines: 3,
|
||||
},
|
||||
// Two columns, like claude's, measured on a live 0.154.0 answer: the `•`/`›`/`⚠`
|
||||
// markers sit in the gutter, prose continuations sit at 2, and a nested YAML block
|
||||
// the model wrote rendered at 2/4/6/8 for its own 0/2/4/6. Replayed at 100, 120,
|
||||
// 160, 198, 235 and 282 columns the indents were 0, 2, 4, 6 and 8 at every one,
|
||||
// never 1, so the width is not a function of the pane.
|
||||
transcriptGutter: 2,
|
||||
transcript: 'codex-rollout',
|
||||
altScreen: 'strip-full',
|
||||
echo: { policy: 'predict', anchor: { kind: 'cursor' }, predictProfile: 'codex' },
|
||||
@@ -523,7 +652,8 @@ const GEMINI: CliEntry = {
|
||||
id: 'gemini' as CliEntry['id'],
|
||||
label: 'Gemini',
|
||||
shortBadge: 'GM',
|
||||
accent: '#4285f4',
|
||||
// The tab badge / run-mode-dot colour, not the run-button border (see the note above CLAUDE).
|
||||
accent: '#8ab4f8',
|
||||
enabled: true,
|
||||
stock: true,
|
||||
order: 30,
|
||||
@@ -615,7 +745,7 @@ const ANTIGRAVITY: CliEntry = {
|
||||
id: 'antigravity' as CliEntry['id'],
|
||||
label: 'Antigravity',
|
||||
shortBadge: 'AG',
|
||||
accent: '#8b5cf6',
|
||||
accent: '#22d3ee',
|
||||
enabled: true,
|
||||
stock: true,
|
||||
order: 40,
|
||||
@@ -685,7 +815,7 @@ const PI: CliEntry = {
|
||||
id: 'pi' as CliEntry['id'],
|
||||
label: 'Pi',
|
||||
shortBadge: 'PI',
|
||||
accent: '#10b981',
|
||||
accent: '#f472b6',
|
||||
enabled: true,
|
||||
stock: true,
|
||||
order: 50,
|
||||
@@ -809,10 +939,10 @@ const GROK: CliEntry = {
|
||||
shortBadge: 'GK',
|
||||
// Upstream hand-authored a charcoal GRADIENT across 4+ CSS spots (welcome button, tab
|
||||
// badge, run-mode dot, mobile skin overrides) rather than one flat colour; our registry's
|
||||
// `accent` is a single hex, so this is the closest single value (the run-mode-dot colour,
|
||||
// zinc-400). Nothing reads `accent` yet — the frontend is untouched in this change and
|
||||
// keeps its own hand-authored CSS; the field is here so the entry is complete.
|
||||
accent: '#a1a1aa',
|
||||
// `accent` is a single hex, so this is the closest single value (zinc-300, the run-button
|
||||
// border and tab-badge colour). Nothing reads `accent` yet: the frontend keeps its own
|
||||
// hand-authored CSS; the field is here so the entry is complete.
|
||||
accent: '#d4d4d8',
|
||||
enabled: true,
|
||||
stock: true,
|
||||
order: 70,
|
||||
@@ -944,7 +1074,7 @@ const DEEPSEEK: CliEntry = {
|
||||
id: 'deepseek' as CliEntry['id'],
|
||||
label: 'DeepSeek',
|
||||
shortBadge: 'DS',
|
||||
accent: '#4d6bfe',
|
||||
accent: '#7c93ff',
|
||||
enabled: true,
|
||||
stock: true,
|
||||
order: 80,
|
||||
@@ -1071,15 +1201,28 @@ const DEEPSEEK: CliEntry = {
|
||||
// privilege rather than granting it, and clamping it here was a real regression
|
||||
// (test/deepseek-mode.test.ts) fixed before this shipped.
|
||||
privilegedEnvKeys: ['DSH_PERMISSION_MODE', 'DSH_HOME', 'DEEPSEEK_BASE_URL'],
|
||||
// Web-researched, unverified, partial: reuses the already-existing DEEPSEEK_BASE_URL/
|
||||
// DEEPSEEK_API_KEY keys above. No modelVars — dsh's model is a profile-composition
|
||||
// entry (see `model: { source: 'none' }` above), not an env var, so forcing a specific
|
||||
// model name may not fully work; verify against a real profile before shipping.
|
||||
// Reuses the already-existing DEEPSEEK_BASE_URL/DEEPSEEK_API_KEY keys above. No
|
||||
// modelVars — dsh's model is a profile-composition entry (see `model: { source: 'none'
|
||||
// }` above), not an env var, so forcing a specific model name may not fully work;
|
||||
// verify against a real profile before shipping.
|
||||
//
|
||||
// ⚠️ appendV1Suffix is REQUIRED, not optional-nice-to-have: without it every request
|
||||
// 404s. Confirmed live and by reading dsh's own bundled source
|
||||
// (@deepseek-ai/dsh-llm-deepseek): it builds the request URL as
|
||||
// `${DEEPSEEK_BASE_URL}/chat/completions` with no "/v1" of its own (its real public
|
||||
// API, https://api.deepseek.com, expects the caller's base URL to already carry any
|
||||
// needed prefix), while llama-swap/llama.cpp only serves the OpenAI-conventional
|
||||
// "/v1/chat/completions" — a bare POST to ".../chat/completions" 404s live, and the
|
||||
// 404 reported here originally ("dsh: HTTP_404: DeepSeek API error (HTTP 404)")
|
||||
// matches dsh's own error-message template for exactly this failure. See the
|
||||
// customModelInjection doc comment in cli-registry/types.ts for the full reasoning,
|
||||
// including why claude/gemini must NOT get this.
|
||||
customModelInjection: {
|
||||
kind: 'env',
|
||||
baseUrlVar: 'DEEPSEEK_BASE_URL',
|
||||
apiKeyVar: 'DEEPSEEK_API_KEY',
|
||||
modelVars: [],
|
||||
appendV1Suffix: true,
|
||||
},
|
||||
},
|
||||
overlays: {
|
||||
@@ -1096,7 +1239,7 @@ const OMP: CliEntry = {
|
||||
id: 'omp' as CliEntry['id'],
|
||||
label: 'OMP',
|
||||
shortBadge: 'OM',
|
||||
accent: '#7c9cf5',
|
||||
accent: '#818cf8',
|
||||
enabled: true,
|
||||
stock: true,
|
||||
order: 90,
|
||||
|
||||
@@ -344,7 +344,61 @@ export interface CliCapabilities {
|
||||
promptGlyph: string;
|
||||
/** Source of a regex matching the status line this CLI draws while a turn runs. */
|
||||
workingLine: string;
|
||||
/**
|
||||
* Source of a regex matching the row this CLI draws while work it started in the
|
||||
* background is still running, e.g. Claude's `· 1 monitor ·` footer chip or Codex's
|
||||
* `1 background terminal running · /ps to view`. Capture group 1 is the label Codeman
|
||||
* shows, and the whole match stands in when the pattern declares no group. A CLI that
|
||||
* omits this reports no background work, which is what every CLI did before the field
|
||||
* existed.
|
||||
*/
|
||||
watchingLine?: string;
|
||||
/**
|
||||
* How many rows at the FOOT of the screen that row can appear in, counting non-blank
|
||||
* rows only. Claude writes its chip on the last row and keeps the default; Codex pins
|
||||
* its own above the composer, which puts it third from the bottom, so it declares
|
||||
* more. Keep each number as small as that CLI's layout allows: every extra row is
|
||||
* another row an agent might be able to write, and the label is what silences an
|
||||
* alert. See `watchingLabel()` in `session-activity.ts`.
|
||||
*/
|
||||
watchingLines?: number;
|
||||
/**
|
||||
* Source of a regex matching the row this CLI closes a turn with when it ended that
|
||||
* turn to WAIT for workers it started and will resume on its own once they finish,
|
||||
* e.g. Claude's `✻ Waiting for 1 dynamic workflow to finish`. A pane showing it counts
|
||||
* as working, not idle: nothing is being asked of the user, and the next turn starts
|
||||
* without them.
|
||||
*
|
||||
* Unlike `workingLine` this is never searched across the pane. The CLI prints the row
|
||||
* once and never updates it, so the copy from an earlier turn is still on screen after
|
||||
* the workers are done. Only the newest transcript row directly above the composer is
|
||||
* tested. See `isAwaitingWorkers()` in `session-activity.ts`.
|
||||
*/
|
||||
awaitingLine?: string;
|
||||
};
|
||||
/**
|
||||
* How many columns this CLI indents its transcript body by, so a copy taken from its
|
||||
* pane can drop that much and paste flush. Claude Code indents two and puts its own
|
||||
* markers in those columns.
|
||||
*
|
||||
* ⚠ DECLARED rather than measured off the pane, and two measured attempts are why.
|
||||
* Asking whether the pane painted real spaces across the unused part of each row
|
||||
* separates a TUI from a shell perfectly where it fires and never over-stripped; it
|
||||
* is also a function of pane WIDTH, because that padding exists only while a
|
||||
* rendered line stops short of the CLI's own layout width and Claude Code's prose
|
||||
* wraps to fill it. On one live transcript the share of padded rows ran 44%, 6%, 6%,
|
||||
* 7% and 87% at 123, 160, 198, 235 and 298 columns, so at any ordinary window size
|
||||
* the strip silently did nothing. Taking the narrowest indent on the surrounding
|
||||
* rows instead fires at every width and over-strips on roughly 1% of selections,
|
||||
* because a file listing inside the transcript can be the narrowest thing on screen.
|
||||
*
|
||||
* A declared width can do neither. The strip is the lesser of this and what every
|
||||
* selected line shares, so a block can only ever shift as a unit, and it can never
|
||||
* shift further than the CLI itself says its gutter is.
|
||||
*
|
||||
* Absent means no strip at all, the same fail-safe direction `workDetect` takes.
|
||||
*/
|
||||
transcriptGutter?: number;
|
||||
/** No direct-PTY fallback: the CLI must run inside tmux (secrets ride tmux setenv). */
|
||||
requiresMux: boolean;
|
||||
/**
|
||||
@@ -496,9 +550,74 @@ export interface CliCapabilities {
|
||||
* declares). Absent = the config alone selects the model (claude's env vars,
|
||||
* opencode's blob, codex's top-level `model` key). Applied by the session's
|
||||
* respawn options through the entry's `legacyConfigField`, never by id.
|
||||
*
|
||||
* `contextLengthVar` (env kind only): the env var a discovered per-model context-window
|
||||
* size is written to when known (claude's `CLAUDE_CODE_MAX_CONTEXT_TOKENS`) — without it,
|
||||
* a CLI that assumes a large default window for an unrecognized model name keeps sending
|
||||
* full-size prompts against a much smaller local server and eventually overflows its real
|
||||
* context (verified: a 33.7K-token system prompt against a 16384-token llama-swap model).
|
||||
* Absent when the CLI has no such override, or the value is unknown for this model.
|
||||
*
|
||||
* `configDirVar` (env kind only): the env var that redirects this session's config/
|
||||
* credential directory to an isolated, per-session one (claude's `CLAUDE_CONFIG_DIR`), so
|
||||
* an injected API key never coexists with a stored claude.ai OAuth session in the same
|
||||
* directory — the CLI still warns "both claude.ai and ANTHROPIC_API_KEY set" when they
|
||||
* share a directory even though the API key wins for actual requests. Isolating it trades
|
||||
* that cosmetic warning for a documented side effect: a relocated config directory writes
|
||||
* transcripts outside `~/.claude/projects`, blinding the response viewer, subagent
|
||||
* windows, and Read My Mind for that session (see docs/wiki/Agent-CLIs.md).
|
||||
*
|
||||
* `apiKeyTrustFile` (env kind only, alongside configDirVar): an isolated config directory
|
||||
* has none of a real profile's prior "detected a custom API key, use it?" approvals, so
|
||||
* without this the CLI stops and asks interactively on every single launch — with no one
|
||||
* at a TTY to answer, that's a hang, not a warning (confirmed live: claude's own default
|
||||
* answer, "No", would silently refuse to use the very key this feature just injected).
|
||||
* `relPath`/`shape` name the file (claude's `.claude.json`) and its
|
||||
* `customApiKeyResponses.approved` field this pre-seeds — the exact field a real answered
|
||||
* prompt itself writes to, so this isn't bypassing the check, just answering it the same
|
||||
* way a one-off prior approval on a shared profile already would.
|
||||
*
|
||||
* `skipFirstRunPrompts` (env kind only, alongside apiKeyTrustFile): an isolated config
|
||||
* directory is not just missing API-key approvals — it is a brand-new profile as far as
|
||||
* the CLI is concerned, so it also replays its ENTIRE first-run sequence on every launch:
|
||||
* the theme picker, the security-notes screen, the per-project "trust this folder?"
|
||||
* dialog, and (running with a bypass-permissions flag) a one-time warning about it —
|
||||
* confirmed live, none of which a real, long-used profile ever shows again. `true`
|
||||
* pre-seeds the same state a real profile accumulates from having answered all of that
|
||||
* once: `hasCompletedOnboarding` and the launching session's own project entry in the
|
||||
* `apiKeyTrustFile` (claude's `.claude.json`), plus `skipDangerousModePermissionPrompt`
|
||||
* in claude's `settings.json` — see `seedFirstRunState`/`seedSkipBypassPermissionsPrompt`
|
||||
* in custom-model-injection-apply.ts. Requires `apiKeyTrustFile` to be set too, since it
|
||||
* reuses that file.
|
||||
*
|
||||
* `appendV1Suffix` (env kind only): the raw `endpoint.baseUrl` gets `withV1Suffix()`
|
||||
* applied before being written to `baseUrlVar`, instead of being used verbatim.
|
||||
* DeepSeek needs this and claude/gemini must NOT get it — a per-CLI asymmetry confirmed
|
||||
* by reading each SDK's own request-building source, not assumed: DeepSeek Harness's
|
||||
* bundled `@deepseek-ai/dsh-llm-deepseek` concatenates `${connection.baseURL}/chat/
|
||||
* completions` with no `/v1` insertion of its own (its real public API base,
|
||||
* `https://api.deepseek.com`, expects the caller's base URL to already carry any
|
||||
* needed prefix), while llama-swap/llama.cpp only ever serves the OpenAI-conventional
|
||||
* `/v1/chat/completions` — confirmed live: a bare `POST <baseUrl>/chat/completions`
|
||||
* 404s, `POST <baseUrl>/v1/chat/completions` succeeds, and the harness's own error
|
||||
* message template (`DeepSeek API error (HTTP ${status})`) reproduces the exact
|
||||
* `HTTP_404` this feature originally shipped with unexplained. Claude Code's own SDK,
|
||||
* by contrast, was already confirmed working end-to-end against the RAW `baseUrl` with
|
||||
* no suffix — appending one there would be wrong, not just redundant.
|
||||
*/
|
||||
customModelInjection:
|
||||
| { kind: 'env'; baseUrlVar: string; apiKeyVar: string; modelVars: string[]; launchModel?: string }
|
||||
| {
|
||||
kind: 'env';
|
||||
baseUrlVar: string;
|
||||
apiKeyVar: string;
|
||||
modelVars: string[];
|
||||
launchModel?: string;
|
||||
contextLengthVar?: string;
|
||||
apiKeyTrustFile?: { relPath: string; shape: 'claude-api-key-responses' };
|
||||
configDirVar?: string;
|
||||
skipFirstRunPrompts?: boolean;
|
||||
appendV1Suffix?: boolean;
|
||||
}
|
||||
| { kind: 'configContentEnv'; envVar: string; template: 'opencode-json'; launchModel?: string }
|
||||
| {
|
||||
kind: 'configDir';
|
||||
@@ -564,12 +683,14 @@ export interface CliOverlays {
|
||||
/**
|
||||
* ⚠️ DECLARED-FOR-LATER: fields no code reads yet.
|
||||
*
|
||||
* `shortBadge`, `accent`, `overlays.credStore`, `capabilities.echo`, `capabilities.wheelForward`,
|
||||
* `accent`, `overlays.credStore`, `capabilities.echo`, `capabilities.wheelForward`,
|
||||
* `capabilities.keyboardAccessory` and `capabilities.maxFrameBytes` all describe FRONTEND
|
||||
* behaviour, and the frontend is deliberately untouched by the change that introduced this
|
||||
* registry — `app.js`, `terminal-ui.js`, `styles.css` and friends keep their own
|
||||
* behaviour, and most of the frontend is deliberately untouched by the change that introduced
|
||||
* this registry — `app.js`, `terminal-ui.js`, `styles.css` and friends keep their own
|
||||
* hand-authored per-CLI rules, and moving them is its own piece of work with its own way of
|
||||
* being verified (a mobile/browser suite the CI gate cannot see).
|
||||
* being verified (a mobile/browser suite the CI gate cannot see). `shortBadge` graduated out of
|
||||
* this list (docs/cli-enable-disable-plan.md, Phase 2): `GET /api/clis` reads it for the
|
||||
* CLI-management Settings list.
|
||||
*
|
||||
* They are declared now because each entry should describe its CLI completely, and because
|
||||
* transcribing them while the hand-written source is still on screen is when the values are
|
||||
@@ -586,7 +707,13 @@ export interface CliEntry {
|
||||
label: string;
|
||||
/** Two-ish character tab badge, e.g. 'OC'. */
|
||||
shortBadge: string;
|
||||
/** Single hex colour. CSS derives every per-CLI gradient from it via --cli-accent. */
|
||||
/**
|
||||
* Single hex colour, measured from the CLI's actual `.btn-toolbar.btn-run.mode-<id>`
|
||||
* gradient in styles.css (see stock.ts's comment above `CLAUDE` for the exact
|
||||
* methodology). DECLARED-FOR-LATER (above) — no code reads this yet; styles.css's
|
||||
* gradients are still hand-authored per id, not derived from this field via any
|
||||
* CSS custom property. There is no `--cli-accent` variable in the codebase.
|
||||
*/
|
||||
accent: string;
|
||||
enabled: boolean;
|
||||
/** Set by the loader from the shipped catalog; a user entry can never claim it. */
|
||||
|
||||
@@ -69,7 +69,9 @@ const ALL: ProbeEnvironment[] = ['linux', 'darwin', 'wsl', 'win32'];
|
||||
* shown to the user and claude's does not follow the pattern.
|
||||
*/
|
||||
const DOCTOR_ROW_OVERRIDES: Record<string, { id?: string; label?: string; usedBy: string[] }> = {
|
||||
claude: { usedBy: ['Claude Code sessions (default backend)'] },
|
||||
// The label override keeps the doctor row's historical "Claude CLI" spelling now that
|
||||
// the registry label is the product name, "Claude Code".
|
||||
claude: { label: 'Claude CLI', usedBy: ['Claude Code sessions (default backend)'] },
|
||||
opencode: { usedBy: ['OpenCode sessions'] },
|
||||
codex: { usedBy: ['Codex sessions'] },
|
||||
gemini: { usedBy: ['Gemini sessions'] },
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
/**
|
||||
* @fileoverview Limits shared between Wake-on-LAN parsing and its request schema.
|
||||
*
|
||||
* Its own module because `src/remote-wake.ts` is import-fenced: only
|
||||
* `web/routes/session-routes.ts` and `web/server.ts` may import it, so that no
|
||||
* watcher or boot-recovery path can WAKE a host (pinned by the wiring guard in
|
||||
* `test/remote-wake.test.ts`). `web/schemas.ts` needs the same MAC-count limit and
|
||||
* must not become a third importer, and it would drag `dgram`/`net`/`child_process`
|
||||
* into every module that validates a request body. A plain constant satisfies both.
|
||||
*/
|
||||
|
||||
/**
|
||||
* How many comma-separated MACs one `wakeMac` may carry.
|
||||
*
|
||||
* ⚠ Single source for `parseMacList()` and `RemoteHostSchema.wakeMac`. The two used
|
||||
* to disagree: the schema's 128-character cap admits seven MACs while the parser
|
||||
* rejected more than four all-or-nothing, so a five-MAC value validated, persisted to
|
||||
* `remote-hosts.json`, and then resolved to NO wake target. The host read as
|
||||
* unconfigured and the banner offered "Configure WoL" for a host the user had just
|
||||
* configured, which is the worst shape a validation gap can take: accepted, stored,
|
||||
* silently inert.
|
||||
*/
|
||||
export const MAX_WAKE_MACS = 4;
|
||||
@@ -99,3 +99,11 @@ export const STALE_DATA_MAX_AGE_MS = 60 * 60 * 1000;
|
||||
|
||||
/** Standard 5-minute inactivity timeout for streams and caches (ms) */
|
||||
export const INACTIVITY_TIMEOUT_MS = 5 * 60 * 1000;
|
||||
|
||||
/**
|
||||
* Gap between a paste-mode cron prompt's text and its Enter (ms). The two must be
|
||||
* separate writes: Claude Code takes a raw `<text>\r` burst of about a hundred
|
||||
* characters as a paste and turns its `\r` into a newline. A separate `\r` 80 ms
|
||||
* after the text was measured to submit; this leaves room for a longer prompt.
|
||||
*/
|
||||
export const CRON_PASTE_ENTER_DELAY_MS = 300;
|
||||
|
||||
@@ -20,7 +20,7 @@ import { getErrorMessage, createErrorResponse, ApiErrorCode } from '../types/api
|
||||
import { MAX_CONCURRENT_SESSIONS, MAX_CRON_JOBS, MAX_CRON_RUN_HISTORY } from '../config/map-limits.js';
|
||||
import { canUsernameRunPrivilegedCommands, resolveClaudeModeForUsername } from '../user-store.js';
|
||||
import { sessionCapacityState, isWorkingDirAllowedForUsername } from '../web/route-helpers.js';
|
||||
import { CRON_READY_MAX_ATTEMPTS, CRON_READY_SETTLE_MS } from '../config/server-timing.js';
|
||||
import { CRON_PASTE_ENTER_DELAY_MS, CRON_READY_MAX_ATTEMPTS, CRON_READY_SETTLE_MS } from '../config/server-timing.js';
|
||||
import {
|
||||
DEFAULT_BLOCKED_TREES,
|
||||
isBlockedAttachmentPath,
|
||||
@@ -100,6 +100,41 @@ const CRON_WORKING_DIR_BLOCKED_TREES: readonly string[] = [...DEFAULT_BLOCKED_TR
|
||||
/** Prompt delivery is single-line only (writeViaMux/Ink constraint). */
|
||||
const HAS_NEWLINE = /[\r\n]/;
|
||||
|
||||
/** The three session calls prompt delivery needs, so it can be tested without a PTY. */
|
||||
type CronPromptTarget = Pick<Session, 'write' | 'writeViaMux' | 'verifySubmitted'>;
|
||||
|
||||
/**
|
||||
* Send a cron job's (single-line) prompt into its session and press Enter.
|
||||
*
|
||||
* `typed` goes through the mux: the text is typed, Enter is its own key, and the
|
||||
* session re-presses it while the prompt is still on the composer.
|
||||
*
|
||||
* `paste` writes the text straight into the PTY, and must send its Enter as a
|
||||
* SEPARATE write. It used to send `<text>\r` in one piece, and Claude Code (measured
|
||||
* on 2.1.283) takes a burst of about a hundred characters as a paste, so the `\r`
|
||||
* landed as a newline and the prompt sat unsent while the run reported
|
||||
* `prompt_sent`. The Enter goes down the same PTY as the text, so it cannot overtake
|
||||
* it, and the same composer check then covers a CLI that was not taking Enter yet.
|
||||
*
|
||||
* @returns false when the session had no PTY or mux to write to
|
||||
*/
|
||||
export async function deliverCronPrompt(
|
||||
target: CronPromptTarget,
|
||||
prompt: string,
|
||||
inputMode: CronJob['inputMode'],
|
||||
wait: (ms: number) => Promise<void> = delay
|
||||
): Promise<boolean> {
|
||||
if (inputMode !== 'paste') {
|
||||
return target.writeViaMux(prompt.endsWith('\r') ? prompt : `${prompt}\r`);
|
||||
}
|
||||
const text = prompt.replace(/[\r\n]+$/, '');
|
||||
if (!target.write(text)) return false;
|
||||
await wait(CRON_PASTE_ENTER_DELAY_MS);
|
||||
if (!target.write('\r')) return false;
|
||||
target.verifySubmitted(text);
|
||||
return true;
|
||||
}
|
||||
|
||||
/** Order-insensitive equality for the weekly-days arrays. */
|
||||
function sameDays(a: number[] | undefined, b: number[] | undefined): boolean {
|
||||
const x = [...(a ?? [])].sort((p, q) => p - q);
|
||||
@@ -623,15 +658,9 @@ export class CronService {
|
||||
const s = this.deps.sessions.get(sessionId);
|
||||
if (!s) return;
|
||||
try {
|
||||
const payload = prompt.endsWith('\r') ? prompt : `${prompt}\r`;
|
||||
let delivered = true;
|
||||
if (job.inputMode === 'paste') {
|
||||
s.write(payload);
|
||||
} else {
|
||||
delivered = await s.writeViaMux(payload);
|
||||
}
|
||||
const delivered = await deliverCronPrompt(s, prompt, job.inputMode);
|
||||
if (!delivered) {
|
||||
this.failRun(job, run, 'Failed to send prompt: mux write failed');
|
||||
this.failRun(job, run, 'Failed to send prompt: the session could not be written to');
|
||||
return;
|
||||
}
|
||||
run.status = 'prompt_sent';
|
||||
|
||||
@@ -40,6 +40,38 @@ export interface CustomModelHost {
|
||||
authStyle?: CustomModelAuthStyle;
|
||||
models?: string[];
|
||||
lastDiscoveredAt?: string;
|
||||
/**
|
||||
* The model the Run-menu picker (docs/custom-model-endpoints-plan.md) applies when
|
||||
* this endpoint is picked with no further choice — one generated menu entry per
|
||||
* (CLI, endpoint) pair, not per (CLI, endpoint, model), so it needs a single answer.
|
||||
* Must be a member of `models` when set; the picker falls back to `models[0]` when
|
||||
* this is unset, and disables the entry entirely when `models` is empty (nothing to
|
||||
* default to). Never auto-set on discovery — the previous default staying valid
|
||||
* after a re-discover is a property worth keeping even if the model list changes.
|
||||
*/
|
||||
defaultModelId?: string;
|
||||
/**
|
||||
* Discovered context-window size (tokens) per model id, keyed by the same strings as
|
||||
* `models`. Populated opportunistically during discovery (`custom-model-routes.ts`) from
|
||||
* llama.cpp/llama-swap's `GET /props?model=<id>` — the plain OpenAI-shaped `/v1/models`
|
||||
* response has no such field. Only ever probed for a model the server already reports as
|
||||
* loaded (llama-swap's `status.value === 'loaded'`); an unloaded one is deliberately never
|
||||
* probed, since llama-swap treats `/props?model=` as a routing hint that can trigger an
|
||||
* actual (slow, GPU-swapping) model load as a side effect of merely asking. A model this
|
||||
* has no entry for simply gets no context-length env override applied — never a guess.
|
||||
*/
|
||||
modelContextLengths?: Record<string, number>;
|
||||
/**
|
||||
* Discovered file size (GB) per model id, keyed by the same strings as `models`.
|
||||
* Populated during discovery by parsing llama-swap's own `description` field for an
|
||||
* auto-discovered model ("Auto-discovered 16.35 GB - parameters auto-fitted by
|
||||
* llama.cpp") — a hand-configured profile's own description has no such figure and
|
||||
* correctly gets no entry, never a guess. Used only to label the Run-menu picker's
|
||||
* "loading model" banner with a rough, unmeasured expected-time estimate
|
||||
* (the Run-menu picker's loading banner in session-ui.js) — never a guarantee, and never anything a
|
||||
* server-side check relies on.
|
||||
*/
|
||||
modelSizesGB?: Record<string, number>;
|
||||
}
|
||||
|
||||
export function customModelHostsPath(configDir: string): string {
|
||||
|
||||
@@ -10,7 +10,8 @@
|
||||
* cli-registry changes" requirement it was written against.
|
||||
*/
|
||||
|
||||
import { chmodSync, mkdirSync, writeFileSync, rmSync } from 'node:fs';
|
||||
import { chmodSync, existsSync, mkdirSync, readFileSync, writeFileSync, rmSync, symlinkSync } from 'node:fs';
|
||||
import { homedir, platform } from 'node:os';
|
||||
import { join, dirname } from 'node:path';
|
||||
import { dataPath } from './config/instance.js';
|
||||
import type { CliEntry } from './config/cli-registry/types.js';
|
||||
@@ -48,6 +49,168 @@ export function applyConfigDirInjection(baseDir: string, injection: ConfigDirInj
|
||||
return { [injection.dirEnvVar]: baseDir, ...injection.extraEnv };
|
||||
}
|
||||
|
||||
/**
|
||||
* Real, shared Claude config directory Codeman's own host process runs under — honors
|
||||
* `CLAUDE_CONFIG_DIR` the same way `claude-credentials.ts`'s `claudeCredentialsPath()`
|
||||
* does, so the symlink below points at wherever `~/.claude/projects` actually lives
|
||||
* rather than assuming the plain default.
|
||||
*/
|
||||
function realClaudeConfigDir(): string {
|
||||
const configured = typeof process.env.CLAUDE_CONFIG_DIR === 'string' && process.env.CLAUDE_CONFIG_DIR.trim();
|
||||
return configured || join(homedir(), '.claude');
|
||||
}
|
||||
|
||||
/**
|
||||
* Symlinks `<isolatedDir>/projects` back to the real, shared `~/.claude/projects`, so an
|
||||
* isolated `CLAUDE_CONFIG_DIR` (used to keep an injected API key away from a stored OAuth
|
||||
* session — see `configDirVar` on customModelInjection) doesn't also blind the response
|
||||
* viewer, subagent windows, and Read My Mind for that session (docs/wiki/Agent-CLIs.md).
|
||||
* Best-effort: a platform that refuses symlinks (unprivileged Windows without a junction
|
||||
* fallback working, e.g.) just keeps the pre-existing documented side effect instead of
|
||||
* failing the whole custom-model apply over a nice-to-have.
|
||||
*/
|
||||
function linkSharedProjectsDir(isolatedDir: string): void {
|
||||
const link = join(isolatedDir, 'projects');
|
||||
if (existsSync(link)) return; // already linked (idempotent re-apply) or real dir wrote one
|
||||
try {
|
||||
symlinkSync(join(realClaudeConfigDir(), 'projects'), link, platform() === 'win32' ? 'junction' : 'dir');
|
||||
} catch {
|
||||
// best-effort only — response viewer/subagent windows go blind for this session instead
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Pre-approves the injected API key in an isolated config directory's trust-dialog state
|
||||
* (`customModelInjection.apiKeyTrustFile`), so an otherwise-empty directory doesn't make the
|
||||
* CLI stop at an interactive "Detected a custom API key — use it?" prompt on every single
|
||||
* launch. Confirmed live: with nobody at the TTY to answer, that prompt's own default
|
||||
* ("No") silently refuses the very key this feature just injected — this isn't bypassing
|
||||
* the check, it's answering it the same field a real answered prompt itself writes to
|
||||
* (verified against a real `~/.claude.json` after answering by hand once).
|
||||
*
|
||||
* Merges rather than overwrites: the file may already carry fields the CLI itself wrote on
|
||||
* an earlier launch in this same isolated directory (machineID, userID, other approved
|
||||
* keys), and a corrupt or partially-written file (a crash mid-write) is treated as absent
|
||||
* rather than failing the whole apply over a nice-to-have.
|
||||
*/
|
||||
/**
|
||||
* The form Claude Code actually stores an approved key in: the trimmed last 20
|
||||
* characters. Mirrors the CLI's own `e.trim().slice(-20)`, which is applied on BOTH
|
||||
* the write and the lookup, so anything else never matches.
|
||||
*/
|
||||
export function truncateApiKeyForTrustFile(apiKey: string): string {
|
||||
return apiKey.trim().slice(-20);
|
||||
}
|
||||
|
||||
function seedApiKeyTrustFile(
|
||||
configDir: string,
|
||||
trustFile: { relPath: string; shape: 'claude-api-key-responses' },
|
||||
apiKey: string
|
||||
): void {
|
||||
const filePath = join(configDir, trustFile.relPath);
|
||||
let existing: Record<string, unknown> = {};
|
||||
try {
|
||||
existing = JSON.parse(readFileSync(filePath, 'utf8')) as Record<string, unknown>;
|
||||
} catch {
|
||||
existing = {};
|
||||
}
|
||||
const responses = (existing.customApiKeyResponses ?? {}) as { approved?: unknown; rejected?: unknown };
|
||||
const approved = new Set(Array.isArray(responses.approved) ? (responses.approved as string[]) : []);
|
||||
// ⚠ Claude Code stores and compares only the LAST 20 CHARACTERS of a key, never the
|
||||
// whole thing: its lookup is `approved.includes(key.trim().slice(-20))` (decompiled
|
||||
// from the 2.1.278 bundle, and corroborated by real `~/.claude.json` files, whose
|
||||
// customApiKeyResponses entries are all exactly 20 characters). Seeding the full key
|
||||
// therefore never matches for a REAL key, and claude stops at the interactive
|
||||
// "Detected a custom API key in your environment" prompt, whose default is
|
||||
// "No (recommended)" — so the launch hangs or silently refuses the key this feature
|
||||
// just injected. It went unnoticed because a keyless llama.cpp/llama-swap endpoint
|
||||
// uses DEFAULT_API_KEY ('local-dummy-key', 15 chars), where slice(-20) is the whole
|
||||
// string and the seed matches by accident. Truncating here also keeps a full
|
||||
// third-party credential from being written into a second file on disk.
|
||||
approved.add(truncateApiKeyForTrustFile(apiKey));
|
||||
const rejected = Array.isArray(responses.rejected) ? responses.rejected : [];
|
||||
existing.customApiKeyResponses = { approved: [...approved], rejected };
|
||||
try {
|
||||
writeFileSync(filePath, JSON.stringify(existing, null, 2), { encoding: 'utf8', mode: 0o600 });
|
||||
chmodSync(filePath, 0o600);
|
||||
} catch {
|
||||
// best-effort only — the interactive prompt returns instead of a hard failure here
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Pre-seeds the two remaining pieces of "already been onboarded" state a fresh
|
||||
* `CLAUDE_CONFIG_DIR` has none of (`customModelInjection.skipFirstRunPrompts`, alongside
|
||||
* apiKeyTrustFile): claude replays its whole first-run sequence — the theme picker, the
|
||||
* security-notes screen, and (per-project) the "trust this folder?" dialog — against ANY
|
||||
* config directory that has never completed it, confirmed live against a genuinely fresh
|
||||
* isolated directory. `hasCompletedOnboarding` skips the theme/security-notes screens
|
||||
* outright; `projects[workingDir].hasTrustDialogAccepted` answers the trust dialog for
|
||||
* THIS session's own working directory the same way a real profile's own prior approval
|
||||
* would — other projects in the file are left alone, and `workingDir` is used verbatim
|
||||
* (never realpath'd or slash-normalized) since that's the literal string claude itself
|
||||
* uses as the project key, being whatever string the session was actually launched with
|
||||
* as its cwd.
|
||||
*
|
||||
* Same merge-not-overwrite and corrupt-file-tolerant behavior as `seedApiKeyTrustFile`
|
||||
* (same file, so a second sequential read-modify-write here is deliberate rather than
|
||||
* folding both into one pass — keeps each seed independently testable and optional).
|
||||
*/
|
||||
function seedFirstRunOnboardingState(
|
||||
configDir: string,
|
||||
trustFile: { relPath: string; shape: 'claude-api-key-responses' },
|
||||
workingDir: string
|
||||
): void {
|
||||
const filePath = join(configDir, trustFile.relPath);
|
||||
let existing: Record<string, unknown> = {};
|
||||
try {
|
||||
existing = JSON.parse(readFileSync(filePath, 'utf8')) as Record<string, unknown>;
|
||||
} catch {
|
||||
existing = {};
|
||||
}
|
||||
existing.hasCompletedOnboarding = true;
|
||||
const projects =
|
||||
existing.projects && typeof existing.projects === 'object' && !Array.isArray(existing.projects)
|
||||
? (existing.projects as Record<string, Record<string, unknown>>)
|
||||
: {};
|
||||
const existingProject = projects[workingDir] && typeof projects[workingDir] === 'object' ? projects[workingDir] : {};
|
||||
projects[workingDir] = { ...existingProject, hasTrustDialogAccepted: true };
|
||||
existing.projects = projects;
|
||||
try {
|
||||
writeFileSync(filePath, JSON.stringify(existing, null, 2), { encoding: 'utf8', mode: 0o600 });
|
||||
chmodSync(filePath, 0o600);
|
||||
} catch {
|
||||
// best-effort only — the interactive dialogs return instead of a hard failure here
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Pre-seeds the "skip the bypass-permissions warning" setting (`customModelInjection.
|
||||
* skipFirstRunPrompts`, alongside apiKeyTrustFile) into an isolated config directory's
|
||||
* `settings.json` — a real, already-onboarded profile answers claude's one-time warning
|
||||
* about running with a bypass-permissions flag once and never sees it again, but every
|
||||
* custom-model session launches with a fresh, otherwise-empty CLAUDE_CONFIG_DIR that
|
||||
* carries none of that (confirmed live). A different file from apiKeyTrustFile's
|
||||
* `.claude.json` — this is claude's own global `settings.json`, not project-keyed —
|
||||
* so it gets its own merge-not-overwrite read-modify-write.
|
||||
*/
|
||||
function seedSkipBypassPermissionsPrompt(configDir: string): void {
|
||||
const filePath = join(configDir, 'settings.json');
|
||||
let existing: Record<string, unknown> = {};
|
||||
try {
|
||||
existing = JSON.parse(readFileSync(filePath, 'utf8')) as Record<string, unknown>;
|
||||
} catch {
|
||||
existing = {};
|
||||
}
|
||||
existing.skipDangerousModePermissionPrompt = true;
|
||||
try {
|
||||
writeFileSync(filePath, JSON.stringify(existing, null, 2), { encoding: 'utf8', mode: 0o600 });
|
||||
chmodSync(filePath, 0o600);
|
||||
} catch {
|
||||
// best-effort only — the interactive warning returns instead of a hard failure here
|
||||
}
|
||||
}
|
||||
|
||||
/** Best-effort recursive removal of a previously-written configDir. Never throws. */
|
||||
export function removeConfigDir(dir: string | undefined): void {
|
||||
if (!dir) return;
|
||||
@@ -79,14 +242,43 @@ export function applyCustomModelInjection(
|
||||
entry: Pick<CliEntry, 'capabilities'>,
|
||||
endpoint: CustomModelEndpoint,
|
||||
modelId: string,
|
||||
sessionId: string
|
||||
sessionId: string,
|
||||
/** Discovered context-window size for `modelId`, if known — see `contextLengthVar`. */
|
||||
contextLength?: number,
|
||||
/**
|
||||
* The session's own working directory — only used for `skipFirstRunPrompts`'s per-project
|
||||
* trust-dialog seed, and only when provided (boot recovery, which has no reason to
|
||||
* re-answer a dialog that already fired once, omits it rather than re-deriving it).
|
||||
*/
|
||||
workingDir?: string
|
||||
): AppliedCustomModel | undefined {
|
||||
const injection = buildCustomModelInjection(entry, endpoint, modelId);
|
||||
const injection = buildCustomModelInjection(entry, endpoint, modelId, contextLength);
|
||||
if (injection.kind === 'unsupported') return undefined;
|
||||
if (injection.kind === 'env') {
|
||||
// `configDirVar` (claude's CLAUDE_CONFIG_DIR): point it at the same isolated,
|
||||
// per-session directory the `configDir` kind uses, but write no files into it — an
|
||||
// empty directory has no stored OAuth credential to conflict with the injected API
|
||||
// key, which is the whole point. Reusing the same path keyed by sessionId keeps this
|
||||
// idempotent across a boot-recovery re-apply, same as the configDir kind below.
|
||||
let envOverrides = injection.envOverrides;
|
||||
let configDir: string | undefined;
|
||||
if (injection.configDirVar) {
|
||||
configDir = customModelConfigDir(sessionId);
|
||||
mkdirSync(configDir, { recursive: true, mode: 0o700 });
|
||||
linkSharedProjectsDir(configDir);
|
||||
if (injection.apiKeyTrustFile && injection.apiKey) {
|
||||
seedApiKeyTrustFile(configDir, injection.apiKeyTrustFile, injection.apiKey);
|
||||
}
|
||||
if (injection.skipFirstRunPrompts && injection.apiKeyTrustFile) {
|
||||
if (workingDir) seedFirstRunOnboardingState(configDir, injection.apiKeyTrustFile, workingDir);
|
||||
seedSkipBypassPermissionsPrompt(configDir);
|
||||
}
|
||||
envOverrides = { ...envOverrides, [injection.configDirVar]: configDir };
|
||||
}
|
||||
return {
|
||||
envOverrides: injection.envOverrides,
|
||||
envKeys: Object.keys(injection.envOverrides),
|
||||
envOverrides,
|
||||
envKeys: Object.keys(envOverrides),
|
||||
configDir,
|
||||
launchModel: injection.launchModel,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -16,14 +16,21 @@
|
||||
* shape was rejected by a real codex binary with "invalid type: map,
|
||||
* expected a string" — caught by `scripts/test-local-llm-harnesses.ts`),
|
||||
* but `wire_api = "responses"` is the only value codex still accepts
|
||||
* (support for `"chat"` was dropped in Feb 2026), and a plain OpenAI
|
||||
* Chat-Completions server (llama.cpp, llama-swap, most local setups) does
|
||||
* NOT implement the Responses API — so codex may still fail at the
|
||||
* PROTOCOL level even with a correctly-shaped config file. That gap is
|
||||
* real and current, not a stale warning; see docs/custom-model-endpoints-plan.md. The rest
|
||||
* (gemini/pi/grok/deepseek/omp) have their ONE-SHOT INVOCATION flags
|
||||
* confirmed against real installed binaries' own `--help` output, but
|
||||
* their custom-endpoint env/config conventions remain web-researched,
|
||||
* (support for `"chat"` was dropped in Feb 2026). ⚠️ Re-verified live
|
||||
* against a llama-swap deployment that DOES answer `/v1/responses`: a
|
||||
* plain, no-tool-call turn gets a real reply, but a real tool-call attempt
|
||||
* comes back as `agent_message` TEXT (the tool-call JSON printed as the
|
||||
* answer) rather than a `function_call` item codex would execute —
|
||||
* confirmed via `codex exec --json`'s raw event stream. Tool execution is
|
||||
* what makes codex a coding agent, so this remains not usable for real
|
||||
* work even where plain chat succeeds; see docs/custom-model-endpoints-plan.md
|
||||
* for the full picture (including the harmless `Model metadata ... not
|
||||
* found` warning every custom-endpoint codex session prints — sourced from
|
||||
* a local cache of OpenAI's OWN hosted model catalog that a custom model
|
||||
* can never appear in, confirmed to have no effect on the outcome above).
|
||||
* The rest (gemini/pi/grok/deepseek/omp) have their ONE-SHOT INVOCATION
|
||||
* flags confirmed against real installed binaries' own `--help` output,
|
||||
* but their custom-endpoint env/config conventions remain web-researched,
|
||||
* unverified.
|
||||
*/
|
||||
|
||||
@@ -44,6 +51,20 @@ export interface EnvInjection {
|
||||
envOverrides: Record<string, string>;
|
||||
/** See {@link ConfigDirInjection.launchModel}. */
|
||||
launchModel?: string;
|
||||
/**
|
||||
* Name of the env var the caller should point at an isolated, credential-free config
|
||||
* directory for this session (claude's `CLAUDE_CONFIG_DIR`), from the registry entry's
|
||||
* `customModelInjection.configDirVar`. The actual directory value isn't computed here —
|
||||
* this module is pure and has no sessionId to derive one from — the IO wrapper
|
||||
* (`custom-model-injection-apply.ts`) creates it and adds it to `envOverrides`.
|
||||
*/
|
||||
configDirVar?: string;
|
||||
/** See `customModelInjection.apiKeyTrustFile` — carried through so the IO wrapper can seed it. */
|
||||
apiKeyTrustFile?: { relPath: string; shape: 'claude-api-key-responses' };
|
||||
/** The literal API key value this injection used, for `apiKeyTrustFile` to pre-approve. */
|
||||
apiKey?: string;
|
||||
/** See `customModelInjection.skipFirstRunPrompts` — carried through so the IO wrapper can seed it. */
|
||||
skipFirstRunPrompts?: boolean;
|
||||
}
|
||||
|
||||
export interface ConfigDirInjection {
|
||||
@@ -90,7 +111,9 @@ function quoted(value: string): string {
|
||||
export function buildCustomModelInjection(
|
||||
entry: Pick<CliEntry, 'capabilities'>,
|
||||
endpoint: CustomModelEndpoint,
|
||||
modelId: string
|
||||
modelId: string,
|
||||
/** Discovered context-window size for `modelId`, if known — see `contextLengthVar`. */
|
||||
contextLength?: number
|
||||
): CustomModelInjectionResult {
|
||||
const cap = entry.capabilities.customModelInjection;
|
||||
const apiKey = endpoint.apiKey?.trim() || DEFAULT_API_KEY;
|
||||
@@ -98,11 +121,18 @@ export function buildCustomModelInjection(
|
||||
switch (cap.kind) {
|
||||
case 'env': {
|
||||
const envOverrides: Record<string, string> = {
|
||||
[cap.baseUrlVar]: endpoint.baseUrl,
|
||||
[cap.baseUrlVar]: cap.appendV1Suffix ? withV1Suffix(endpoint.baseUrl) : endpoint.baseUrl,
|
||||
[cap.apiKeyVar]: apiKey,
|
||||
};
|
||||
for (const modelVar of cap.modelVars) envOverrides[modelVar] = modelId;
|
||||
return withLaunchModel({ kind: 'env', envOverrides }, cap.launchModel, modelId);
|
||||
if (cap.contextLengthVar && contextLength !== undefined && Number.isFinite(contextLength)) {
|
||||
envOverrides[cap.contextLengthVar] = String(Math.trunc(contextLength));
|
||||
}
|
||||
let result: EnvInjection = withLaunchModel({ kind: 'env', envOverrides }, cap.launchModel, modelId);
|
||||
if (cap.configDirVar) result = { ...result, configDirVar: cap.configDirVar };
|
||||
if (cap.apiKeyTrustFile) result = { ...result, apiKeyTrustFile: cap.apiKeyTrustFile, apiKey };
|
||||
if (cap.skipFirstRunPrompts) result = { ...result, skipFirstRunPrompts: true };
|
||||
return result;
|
||||
}
|
||||
|
||||
case 'configContentEnv': {
|
||||
@@ -177,11 +207,13 @@ function renderConfigFile(
|
||||
// `env_key`, the NAME of an env var it reads the credential from at runtime, so the
|
||||
// actual value must ride along as an extra env var, never embedded in the file.
|
||||
// ⚠️ `wire_api = "responses"` is the only value codex still accepts (it dropped
|
||||
// `"chat"` support in Feb 2026) — a plain OpenAI Chat-Completions server (llama.cpp,
|
||||
// llama-swap, most local setups) does NOT implement the Responses API, so this
|
||||
// recipe may still fail at the PROTOCOL level even though the file now parses
|
||||
// correctly. That is a real, currently-unresolved compatibility gap, not a syntax
|
||||
// bug — track it before calling codex support done.
|
||||
// `"chat"` support in Feb 2026). Even against a llama-swap deployment that DOES
|
||||
// answer `/v1/responses`, a real tool-call attempt came back as plain TEXT (the
|
||||
// tool-call JSON printed as the model's answer) rather than an executable
|
||||
// `function_call` item — confirmed live via `codex exec --json`. Tool execution is
|
||||
// what makes codex a coding agent, so this remains not usable for real work even
|
||||
// where plain chat succeeds — see the confidence table in
|
||||
// docs/custom-model-endpoints-plan.md, not a syntax bug in this file.
|
||||
const content = [
|
||||
`model = ${quoted(modelId)}`,
|
||||
`model_provider = "custom"`,
|
||||
|
||||
+105
-5
@@ -606,9 +606,65 @@ export function agentImageNpmPackages(): string[] {
|
||||
return packages;
|
||||
}
|
||||
|
||||
/** The `--build-arg` pairs the agent image takes. */
|
||||
export function agentImageBuildArgPairs(): Array<[string, string]> {
|
||||
return [['CLI_NPM_PACKAGES', agentImageNpmPackages().join(' ')]];
|
||||
/**
|
||||
* Environment variable → agent.Dockerfile ARG for the optional git-host CLIs (gh, az).
|
||||
* ⚠️ Mirrors `GIT_HOST_CLI_BUILD_ARGS` in `scripts/lib/cli-catalog.mjs`; the parity test pins them.
|
||||
*/
|
||||
export const GIT_HOST_CLI_BUILD_ARGS: ReadonlyArray<readonly [string, string]> = [
|
||||
['CODEMAN_AGENT_IMAGE_INSTALL_GH', 'CODEMAN_INSTALL_GH'],
|
||||
['CODEMAN_AGENT_IMAGE_INSTALL_AZ', 'CODEMAN_INSTALL_AZ'],
|
||||
];
|
||||
|
||||
/**
|
||||
* Environment variable → Dockerfile ARG for the image's system Git identity.
|
||||
* ⚠️ Mirrors `GIT_IDENTITY_BUILD_ARGS` in `scripts/lib/cli-catalog.mjs`; the parity test pins them.
|
||||
*/
|
||||
export const GIT_IDENTITY_BUILD_ARGS: ReadonlyArray<readonly [string, string]> = [
|
||||
['CODEMAN_AGENT_IMAGE_GIT_USER_NAME', 'GIT_USER_NAME'],
|
||||
['CODEMAN_AGENT_IMAGE_GIT_USER_EMAIL', 'GIT_USER_EMAIL'],
|
||||
];
|
||||
|
||||
/**
|
||||
* The `--build-arg` pairs for the optional git-host CLIs. PURE. An unset or empty variable
|
||||
* contributes NOTHING, so the Dockerfile's own default (off) applies and the argv is the same
|
||||
* as before these existed; anything other than 0/1 is refused rather than guessed at.
|
||||
*/
|
||||
export function gitHostCliBuildArgPairs(env: NodeJS.ProcessEnv): Array<[string, string]> {
|
||||
const pairs: Array<[string, string]> = [];
|
||||
for (const [envName, argName] of GIT_HOST_CLI_BUILD_ARGS) {
|
||||
const value = env[envName];
|
||||
if (value === undefined || value === '') continue;
|
||||
if (value !== '0' && value !== '1') {
|
||||
throw new Error(`${envName} must be 0 or 1, got ${JSON.stringify(value)}`);
|
||||
}
|
||||
pairs.push([argName, value]);
|
||||
}
|
||||
return pairs;
|
||||
}
|
||||
|
||||
/**
|
||||
* The `--build-arg` pairs for a configured Git identity. An absent pair leaves
|
||||
* Git unconfigured, preserving existing deployments; a partial pair is refused.
|
||||
* ⚠️ Mirrors `gitIdentityBuildArgPairs()` in `scripts/lib/cli-catalog.mjs`; the parity test pins them.
|
||||
*/
|
||||
export function gitIdentityBuildArgPairs(env: NodeJS.ProcessEnv): Array<[string, string]> {
|
||||
const pairs = GIT_IDENTITY_BUILD_ARGS.map(([envName, argName]) => [argName, env[envName] ?? ''] as [string, string]);
|
||||
const configured = pairs.filter(([, value]) => value !== '');
|
||||
if (configured.length === 0) return [];
|
||||
if (configured.length !== pairs.length) {
|
||||
const names = GIT_IDENTITY_BUILD_ARGS.map(([envName]) => envName).join(' and ');
|
||||
throw new Error(`${names} must both be set when configuring Git identity`);
|
||||
}
|
||||
return pairs;
|
||||
}
|
||||
|
||||
/** The `--build-arg` pairs the agent image takes. PURE given `env`. */
|
||||
export function agentImageBuildArgPairs(env: NodeJS.ProcessEnv = process.env): Array<[string, string]> {
|
||||
return [
|
||||
['CLI_NPM_PACKAGES', agentImageNpmPackages().join(' ')],
|
||||
...gitHostCliBuildArgPairs(env),
|
||||
...gitIdentityBuildArgPairs(env),
|
||||
];
|
||||
}
|
||||
|
||||
// ========== Credential mount resolution (IO) ==========
|
||||
@@ -785,6 +841,14 @@ interface CredStorePolicy {
|
||||
seedFiles?: string[];
|
||||
/** Seed the WHOLE dir (RO mount → cp -a) — for stores with no shared/host-read state. */
|
||||
seedWhole?: boolean;
|
||||
/**
|
||||
* Seed this store ONLY when this environment variable is exactly `1`, read when
|
||||
* the container is created. For credentials that belong to an opt-in tool rather
|
||||
* than to an agent CLI every case already trusts: they are not inert just because
|
||||
* the image lacks the tool (a gh `hosts.yml` token or an Azure refresh token is
|
||||
* usable by anything in the container, and the agent in it is prompt-injectable).
|
||||
*/
|
||||
enabledByEnv?: string;
|
||||
}
|
||||
|
||||
const CRED_STORES: CredStorePolicy[] = [
|
||||
@@ -829,6 +893,31 @@ const CRED_STORES: CredStorePolicy[] = [
|
||||
},
|
||||
{ rel: '.config/gcloud', seedWhole: true },
|
||||
{ rel: '.config/opencode', seedWhole: true },
|
||||
// GitHub CLI: `hosts.yml` holds the token wherever no system keyring exists (the
|
||||
// Docker server image, a headless Linux host), `config.yml` the preferences. An
|
||||
// agent image built with CODEMAN_INSTALL_GH=1 routes github.com git credentials
|
||||
// through `gh`, so this seed is what lets an agent clone/push a private repo. A
|
||||
// token that lives in a desktop keyring is not in `hosts.yml` and does not carry
|
||||
// in; sign `gh` in inside the container. OPT-IN: seeded only when the same switch
|
||||
// that builds gh into the agent image is on, never merely because the file exists.
|
||||
{ rel: '.config/gh', seedFiles: ['hosts.yml', 'config.yml'], enabledByEnv: 'CODEMAN_AGENT_IMAGE_INSTALL_GH' },
|
||||
// Azure CLI: only the sign-in state. `~/.azure` also accumulates `logs/`,
|
||||
// `commands/`, telemetry and (on a bare host) `cliextensions/`, none of which is
|
||||
// needed to authenticate; the agent image carries its own extensions outside HOME.
|
||||
// `msal_token_cache.json` is plaintext only on Linux (Windows/macOS encrypt it), so
|
||||
// this carries a sign-in from the Docker server image or a Linux host.
|
||||
// OPT-IN like gh: the MSAL cache holds refresh tokens for the whole Azure account.
|
||||
{
|
||||
rel: '.azure',
|
||||
enabledByEnv: 'CODEMAN_AGENT_IMAGE_INSTALL_AZ',
|
||||
seedFiles: [
|
||||
'azureProfile.json',
|
||||
'msal_token_cache.json',
|
||||
'service_principal_entries.json',
|
||||
'clouds.config',
|
||||
'config',
|
||||
],
|
||||
},
|
||||
// OMP keeps its config in `~/.omp/agent` (config.yml/mcp.json/models.yml/
|
||||
// settings.yml — small, no bigger than grok's config.toml/pager.toml), but
|
||||
// that dir ALSO holds agent.db/history.db/models.db (SQLite caches) and
|
||||
@@ -853,10 +942,14 @@ const CRED_STORES: CredStorePolicy[] = [
|
||||
* session state back into the host). Every path is existsSync-gated (on most hosts
|
||||
* only a subset exists). Pure-ish IO (no writes; just existence checks + mount specs).
|
||||
*/
|
||||
export function resolveDockerCredentialArtifacts(home: string = homedir()): DockerClaudeArtifacts {
|
||||
export function resolveDockerCredentialArtifacts(
|
||||
home: string = homedir(),
|
||||
env: NodeJS.ProcessEnv = process.env
|
||||
): DockerClaudeArtifacts {
|
||||
const mounts: DockerMount[] = [];
|
||||
const seedCopies: DockerSeedCopy[] = [];
|
||||
for (const store of CRED_STORES) {
|
||||
if (store.enabledByEnv && env[store.enabledByEnv] !== '1') continue;
|
||||
const hostBase = join(home, store.rel);
|
||||
if (!existsSync(hostBase)) continue;
|
||||
const containerBase = `${CONTAINER_HOME}/${store.rel}`;
|
||||
@@ -1136,10 +1229,17 @@ function buildAgentImage(
|
||||
error: `docker/agent.Dockerfile not found in this install; clone the repo or build ${image} manually`,
|
||||
});
|
||||
}
|
||||
let buildArgPairs: Array<[string, string]>;
|
||||
try {
|
||||
buildArgPairs = agentImageBuildArgPairs();
|
||||
} catch (err) {
|
||||
// A malformed CODEMAN_AGENT_IMAGE_* value: report it like any other build failure.
|
||||
return Promise.resolve({ ok: false, built: false, alreadyPresent: false, error: String((err as Error).message) });
|
||||
}
|
||||
const argv = dockerEngineArgv(docker);
|
||||
const args = [
|
||||
...argv.slice(1),
|
||||
...agentImageBuildArgs(resolved.dockerfile, image, resolved.contextDir, opts.noCache, agentImageBuildArgPairs()),
|
||||
...agentImageBuildArgs(resolved.dockerfile, image, resolved.contextDir, opts.noCache, buildArgPairs),
|
||||
];
|
||||
return new Promise<EnsureImageResult>((resolve) => {
|
||||
// async spawn (NEVER spawnSync) so a multi-minute build never wedges the event loop.
|
||||
|
||||
+30
-6
@@ -195,6 +195,8 @@ export interface CloneOptions {
|
||||
/** `--depth 1`: history-less but much faster on large repos. */
|
||||
shallow?: boolean;
|
||||
timeoutMs?: number;
|
||||
/** Clear every git credential helper for this run (see `GIT_NO_CREDENTIAL_HELPERS`). */
|
||||
withoutCredentialHelpers?: boolean;
|
||||
}
|
||||
|
||||
export type CloneResult = { ok: true; stderr: string } | { ok: false; failure: GitFailure };
|
||||
@@ -436,12 +438,27 @@ export function isSafeGitRef(ref: string): boolean {
|
||||
|
||||
// ─── Pure: argv + env ────────────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Global git options that empty the credential-helper list for one run.
|
||||
*
|
||||
* Every Codeman user in multi-user mode runs git as the SAME OS account, so a
|
||||
* helper that account has (the Docker image's opt-in `gh`/`az` helpers, or a
|
||||
* user's own `gh auth setup-git`) would read private repositories on the
|
||||
* signed-in admin's behalf for anyone who can reach Clone Repo. An empty
|
||||
* `credential.helper` resets the helper list, and a command-line `-c` is read
|
||||
* last, so it also drops the URL-scoped `credential.<url>.helper` entries the
|
||||
* image configures (verified against a real private repo: refs with the helper,
|
||||
* `could not read Username` with it cleared). Public repositories are
|
||||
* unaffected. It must precede the subcommand.
|
||||
*/
|
||||
export const GIT_NO_CREDENTIAL_HELPERS: readonly string[] = ['-c', 'credential.helper='];
|
||||
|
||||
/**
|
||||
* argv for the clone. `--` separates flags from operands so neither the
|
||||
* repository nor the destination can ever be read as an option.
|
||||
*/
|
||||
export function buildCloneArgs(opts: CloneOptions): string[] {
|
||||
const args = ['clone'];
|
||||
const args = [...(opts.withoutCredentialHelpers ? GIT_NO_CREDENTIAL_HELPERS : []), 'clone'];
|
||||
// `--single-branch` is what makes "just this tag/branch" cheap on a big repo.
|
||||
if (opts.ref) args.push('--single-branch', '--branch', opts.ref);
|
||||
if (opts.shallow) args.push('--depth', '1');
|
||||
@@ -450,8 +467,14 @@ export function buildCloneArgs(opts: CloneOptions): string[] {
|
||||
}
|
||||
|
||||
/** argv for the preflight. `--symref` is what reveals the remote's default branch. */
|
||||
export function buildLsRemoteArgs(repository: string): string[] {
|
||||
return ['ls-remote', '--symref', '--', repository];
|
||||
export function buildLsRemoteArgs(repository: string, opts: { withoutCredentialHelpers?: boolean } = {}): string[] {
|
||||
return [
|
||||
...(opts.withoutCredentialHelpers ? GIT_NO_CREDENTIAL_HELPERS : []),
|
||||
'ls-remote',
|
||||
'--symref',
|
||||
'--',
|
||||
repository,
|
||||
];
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -579,7 +602,7 @@ export function classifyGitFailure(stderr: string, timedOut: boolean, spawnError
|
||||
return {
|
||||
code: 'AUTH_REQUIRED',
|
||||
message:
|
||||
'That repository needs authentication. Codeman clones without credentials, so private repositories have to be cloned outside Codeman and added with Link Existing.',
|
||||
"That repository needs authentication. Codeman never asks for credentials, so sign this server's git in first (for example `gh auth login` or `az login` from a shell session; the Docker image can include both, see docker/README.md), or clone it outside Codeman and add it with Link Existing.",
|
||||
stderr: clean,
|
||||
};
|
||||
}
|
||||
@@ -790,7 +813,8 @@ export function isGitAvailable(): boolean {
|
||||
*/
|
||||
export async function probeGitRemote(
|
||||
repository: string,
|
||||
timeoutMs = GIT_LS_REMOTE_TIMEOUT_MS
|
||||
timeoutMs = GIT_LS_REMOTE_TIMEOUT_MS,
|
||||
opts: { withoutCredentialHelpers?: boolean } = {}
|
||||
): Promise<GitRemoteProbe> {
|
||||
if (!isGitAvailable()) {
|
||||
return {
|
||||
@@ -800,7 +824,7 @@ export async function probeGitRemote(
|
||||
failure: classifyGitFailure('', false, 'ENOENT: git not found'),
|
||||
};
|
||||
}
|
||||
const run = await runGit(buildLsRemoteArgs(repository), timeoutMs, MAX_LS_REMOTE_BYTES);
|
||||
const run = await runGit(buildLsRemoteArgs(repository, opts), timeoutMs, MAX_LS_REMOTE_BYTES);
|
||||
if (run.code !== 0 || run.spawnError) {
|
||||
return {
|
||||
reachable: false,
|
||||
|
||||
+69
-1
@@ -24,6 +24,7 @@ import type {
|
||||
OmpConfig,
|
||||
SessionRemote,
|
||||
SessionDocker,
|
||||
PaneExit,
|
||||
} from './types.js';
|
||||
|
||||
/**
|
||||
@@ -56,6 +57,23 @@ export interface MuxSession {
|
||||
respawnConfig?: PersistedRespawnConfig;
|
||||
/** Whether Ralph / Todo tracking is enabled */
|
||||
ralphEnabled?: boolean;
|
||||
/**
|
||||
* This record was rebuilt from the tmux socket rather than from Codeman's own
|
||||
* bookkeeping, so everything on it but the name and the pid is a guess. Its
|
||||
* synthetic `restored-<fragment>` id cannot find the session's `state.json`
|
||||
* entry either, which means a remote or docker session rediscovered this way
|
||||
* arrives with no `remote`/`docker` metadata and looks local. Anything that
|
||||
* would be WRONG about such a session rather than merely vague must fail
|
||||
* closed on this flag.
|
||||
*
|
||||
* ⚠ It is PERMANENT, not merely true for the boot that rediscovered the
|
||||
* session: `saveSessions()` serializes the whole record to
|
||||
* `mux-sessions.json` and `loadSessions()` restores it, so a genuinely local
|
||||
* session rediscovered once stays opted out of everything keyed on this for
|
||||
* the life of that record. That is the safe direction to fail, and it costs
|
||||
* only the guess Codeman is declining to make.
|
||||
*/
|
||||
discovered?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -72,6 +90,13 @@ export interface CreateSessionOptions {
|
||||
workingDir: string;
|
||||
mode: SessionMode;
|
||||
name?: string;
|
||||
/**
|
||||
* Name pinned on a claude spawn as `--name` (version-gated, sanitized, local only).
|
||||
* Deliberately NOT `name`: `--name` owns the prompt-box label, the `/resume` picker
|
||||
* entry and the terminal title, and a pinned title stops Claude generating its own,
|
||||
* so only a user-chosen name belongs here (see `Session.cliPinnedName`).
|
||||
*/
|
||||
cliName?: string;
|
||||
niceConfig?: NiceConfig;
|
||||
model?: string;
|
||||
claudeMode?: ClaudeMode;
|
||||
@@ -105,8 +130,15 @@ export interface RespawnPaneOptions {
|
||||
sessionId: string;
|
||||
workingDir: string;
|
||||
mode: SessionMode;
|
||||
/** Session display name; a respawned claude keeps its `--name` peer name (version-gated, local only). */
|
||||
/** Session display name (tab name). */
|
||||
name?: string;
|
||||
/**
|
||||
* Name pinned on a respawned claude as `--name` (version-gated, sanitized, local only).
|
||||
* Deliberately NOT `name`: `--name` owns the prompt-box label, the `/resume` picker
|
||||
* entry and the terminal title, and a pinned title stops Claude generating its own,
|
||||
* so only a user-chosen name belongs here (see `Session.cliPinnedName`).
|
||||
*/
|
||||
cliName?: string;
|
||||
niceConfig?: NiceConfig;
|
||||
model?: string;
|
||||
claudeMode?: ClaudeMode;
|
||||
@@ -159,6 +191,15 @@ export interface PaneCaptureOptions {
|
||||
* the 1MB execSync default (ENOBUFS).
|
||||
*/
|
||||
maxCaptureBytes?: number;
|
||||
/**
|
||||
* Filled in by the implementation with the pane geometry the capture was
|
||||
* really taken at, which is not always the geometry the caller last asked
|
||||
* for: a resize and a capture can race, and a pane whose size a desktop
|
||||
* viewport has claimed ignores a smaller client's resize outright. A
|
||||
* visible-frame capture addresses every row absolutely, so a consumer
|
||||
* rendering it needs the real height to know the frame fits.
|
||||
*/
|
||||
capturedGeometry?: { cols: number; rows: number };
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -171,6 +212,7 @@ export interface PaneCaptureOptions {
|
||||
* - `sessionKilled` (data: { sessionId: string }) - Session terminated
|
||||
* - `sessionDied` (data: { sessionId: string }) - Session died unexpectedly
|
||||
* - `statsUpdated` (sessions: MuxSessionWithStats[]) - Stats refreshed
|
||||
* - `paneExitsUpdated` () - A pane read finished; ask `getPaneExit()` per session
|
||||
*/
|
||||
export interface TerminalMultiplexer extends EventEmitter {
|
||||
/** Which backend this instance uses */
|
||||
@@ -285,6 +327,32 @@ export interface TerminalMultiplexer extends EventEmitter {
|
||||
/** Check if the pane in a session is dead (command exited but remain-on-exit keeps it alive) */
|
||||
isPaneDead(muxName: string): boolean;
|
||||
|
||||
/**
|
||||
* What the last pane read saw of this session's agent, or `undefined` for
|
||||
* UNKNOWN (Ark0N/Codeman#446). Unlike `isPaneDead()` this costs nothing: it
|
||||
* reads a map the batched watcher fills, so it answers no fresher than that
|
||||
* watcher's interval and the three synchronous `isPaneDead()` callers still
|
||||
* need their own probe. See {@link PaneExit}.
|
||||
*/
|
||||
getPaneExit?(muxName: string): PaneExit | undefined;
|
||||
|
||||
/**
|
||||
* How many authoritative pane reads have agreed on the exit `getPaneExit()`
|
||||
* reports, or 0 when it reports none. The exited-agent sweep closes a session
|
||||
* only once this reaches `CLEAN_EXIT_CONFIRMING_READS` (`pane-exit-sweep.ts`),
|
||||
* and a multiplexer without this method never has a session closed by it.
|
||||
*/
|
||||
getPaneExitReadCount?(muxName: string): number;
|
||||
|
||||
/** Forget a session's exit observation, e.g. once its pane has been respawned. */
|
||||
clearPaneExit?(muxName: string): void;
|
||||
|
||||
/** Start polling every pane on the socket for an exited agent. */
|
||||
startPaneExitWatcher?(intervalMs?: number): void;
|
||||
|
||||
/** Stop the pane-exit watcher. */
|
||||
stopPaneExitWatcher?(): void;
|
||||
|
||||
/** Respawn a dead pane with a fresh command. Returns the new PID or null on failure. */
|
||||
respawnPane(options: RespawnPaneOptions): Promise<number | null>;
|
||||
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
/**
|
||||
* @fileoverview The exited-agent sweep's decision rule (Ark0N/Codeman#446).
|
||||
*
|
||||
* Codeman creates every tmux pane with `remain-on-exit on`, so `/exit` ends the
|
||||
* CLI while the pane, the tmux session and the `tmux attach-session` process
|
||||
* all live on. Part 1 of #446 records that as `SessionState.paneExit`. This
|
||||
* module decides when such a session is closed, the way the X button closes
|
||||
* it, so finished sessions stop piling up on the board.
|
||||
*
|
||||
* The rule closes a session only on a POSITIVE observation of a clean exit:
|
||||
*
|
||||
* - The exit status must be an explicit numeric 0 with no signal. An absent
|
||||
* status is UNKNOWN, never 0: on tmux 3.2a a SIGKILLed pane reports neither a
|
||||
* status nor a signal, so reading absence as clean would sweep an agent the
|
||||
* OOM killer took. A non-zero status or any signal keeps the row, marked with
|
||||
* the exit, as the crash evidence #210 was filed to keep.
|
||||
* - At least {@link CLEAN_EXIT_CONFIRMING_READS} authoritative pane reads must
|
||||
* have agreed on that exit. A failed, empty or skipped read counts for
|
||||
* nothing, because unknown never closes anything.
|
||||
* - No start, attach or relaunch may be in flight for the session. The
|
||||
* dead-pane branch of `Session._setupOrAttachMuxSession()` respawns an exited
|
||||
* pane on purpose, and for a few seconds that pane still reads as dead.
|
||||
* - The exit must land at least {@link CLEAN_EXIT_MIN_PANE_LIFETIME_MS} after
|
||||
* the last start, attach or relaunch finished. A CLI that prints a startup
|
||||
* error ("not logged in", a bad profile, a config error) and exits 0 would
|
||||
* otherwise lose its tab, and the error with it, seconds after launch. Its
|
||||
* row stays, marked `exited (0)`, for the user to read and close.
|
||||
*
|
||||
* Scoping to local mux-backed sessions happens before this rule runs:
|
||||
* `Session.setPaneExit()` forces the field to UNKNOWN for direct-PTY, remote,
|
||||
* docker and discovered sessions, so their `paneExit` never reaches here.
|
||||
*
|
||||
* Pure, so the rule is unit-tested without a server (test/pane-exit-sweep.test.ts).
|
||||
*/
|
||||
import type { PaneExit } from './types/index.js';
|
||||
|
||||
/**
|
||||
* How many authoritative pane reads must agree on a clean exit before the
|
||||
* session is closed. At the watcher's 2 s cadence two reads mean a finished
|
||||
* session disappears within about four seconds of its agent exiting.
|
||||
*/
|
||||
export const CLEAN_EXIT_CONFIRMING_READS = 2;
|
||||
|
||||
/**
|
||||
* How long a pane must have been up before a clean exit closes its session.
|
||||
* An exit sooner than this after the last pane start is read as a startup
|
||||
* failure rather than a user ending the agent, and the row is kept.
|
||||
*/
|
||||
export const CLEAN_EXIT_MIN_PANE_LIFETIME_MS = 10_000;
|
||||
|
||||
/** The lifecycle-log reason recorded when the sweep closes a session. */
|
||||
export const CLEAN_EXIT_CLOSE_REASON = 'agent exited cleanly (status 0)';
|
||||
|
||||
/**
|
||||
* Is this exit a clean one? True only for an explicit numeric status of 0 with
|
||||
* no signal reported.
|
||||
*
|
||||
* ⚠ Never widen this to `(exit.status ?? 0) === 0` or to "no signal, so it was
|
||||
* clean". An absent status is how a signal death presents on tmux 3.2a, and
|
||||
* that shortcut would close crashed agents with nothing failing to warn you.
|
||||
*/
|
||||
export function isCleanPaneExit(exit: PaneExit | undefined): boolean {
|
||||
if (!exit) return false;
|
||||
if (exit.signal !== undefined) return false;
|
||||
return exit.status === 0;
|
||||
}
|
||||
|
||||
/** Everything the sweep needs to know about one session. */
|
||||
export interface CleanExitSweepCandidate {
|
||||
/** The session's published exit, already scoped by `Session.setPaneExit()`. */
|
||||
paneExit: PaneExit | undefined;
|
||||
/** Authoritative pane reads that agreed on that exit (`getPaneExitReadCount()`). */
|
||||
confirmingReads: number;
|
||||
/** A start, attach or relaunch is running for this session's pane. */
|
||||
paneLifecycleInFlight: boolean;
|
||||
/** The session is already being closed or detached. */
|
||||
closing: boolean;
|
||||
/**
|
||||
* When the last start, attach or relaunch of this pane finished
|
||||
* (`Session.paneStartedAt`), or 0 when none has run in this process.
|
||||
*/
|
||||
paneStartedAt: number;
|
||||
}
|
||||
|
||||
/** Should the sweep close this session now? See the file overview for the rule. */
|
||||
export function shouldCloseCleanlyExitedSession(candidate: CleanExitSweepCandidate): boolean {
|
||||
if (candidate.closing) return false;
|
||||
if (candidate.paneLifecycleInFlight) return false;
|
||||
if (!isCleanPaneExit(candidate.paneExit)) return false;
|
||||
// `at` is when this server first read the pane dead, so an exit during the
|
||||
// start itself lands BEFORE `paneStartedAt` and is kept too.
|
||||
if (
|
||||
candidate.paneStartedAt > 0 &&
|
||||
candidate.paneExit!.at - candidate.paneStartedAt < CLEAN_EXIT_MIN_PANE_LIFETIME_MS
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
return candidate.confirmingReads >= CLEAN_EXIT_CONFIRMING_READS;
|
||||
}
|
||||
+27
-14
@@ -19,19 +19,22 @@
|
||||
* touching its status, so a pinned session a reboot killed still reads `idle` or
|
||||
* `busy` and stays eligible.
|
||||
*
|
||||
* ⚠️ Ending the AGENT rather than the session is a shape this module CANNOT
|
||||
* recognise today, and a reboot restores it. `/exit` ends the CLI inside the
|
||||
* pane, `remain-on-exit` keeps the pane, and the PTY Codeman owns is the
|
||||
* `tmux attach-session` process, which stays alive throughout — so no exit
|
||||
* handler runs, no lifecycle `exit` is logged, and the record keeps both its pid
|
||||
* and `status: 'idle'`. Nothing durable distinguishes it from a session that was
|
||||
* simply idle when the power went. Ark0N/Codeman#446 covers making Codeman
|
||||
* notice the dead pane; until a record can say the agent is gone, this pass will
|
||||
* offer those sessions back, and the user dismisses or closes them.
|
||||
* ⚠️ Ending the AGENT rather than the session leaves no trace in `status` or
|
||||
* `pid`. `/exit` ends the CLI inside the pane, `remain-on-exit` keeps the pane,
|
||||
* and the PTY Codeman owns is the `tmux attach-session` process, which stays
|
||||
* alive throughout — so no exit handler runs, no lifecycle `exit` is logged,
|
||||
* and the record keeps both its pid and `status: 'idle'`. Ark0N/Codeman#446
|
||||
* handles it in two steps. The pane-exit watcher persists `paneExit`, and the
|
||||
* clean-exit sweep (`pane-exit-sweep.ts`) closes a session whose agent exited
|
||||
* with status 0 through `cleanupSession()`, which leaves the durable record
|
||||
* described above. This module also refuses a record whose persisted
|
||||
* `paneExit` is a clean exit, which covers a session that exited moments
|
||||
* before the power went, before the sweep reached it. A crashed agent's record
|
||||
* stays eligible, like the row the sweep leaves on the board for it.
|
||||
*
|
||||
* The `pid` check below is therefore NOT that rule. It refuses a record whose
|
||||
* attach process was already gone, which is a session that never started or
|
||||
* whose pane died outright.
|
||||
* The `pid` check below is NOT that rule. It refuses a record whose attach
|
||||
* process was already gone, which is a session that never started or whose
|
||||
* pane died outright.
|
||||
*
|
||||
* @dependencies types (SessionState), config/cli-registry
|
||||
* @consumedby web/server (plan build at boot), web/routes/reboot-restore-routes
|
||||
@@ -41,6 +44,7 @@
|
||||
|
||||
import type { SessionState } from './types.js';
|
||||
import { getCli } from './config/cli-registry/registry.js';
|
||||
import { isCleanPaneExit } from './pane-exit-sweep.js';
|
||||
|
||||
/** Session statuses a reboot restore may rebuild. `stopped` is the kill marker. */
|
||||
const RESTORABLE_STATUSES: ReadonlySet<string> = new Set(['idle', 'busy', 'error']);
|
||||
@@ -109,7 +113,7 @@ export function resolveResumeConversationId(state: SessionState): string {
|
||||
/**
|
||||
* Why one session was passed over. Reported for logging and shown to the user.
|
||||
*
|
||||
* The first seven are decided before anything is built. `capacity-reached` and
|
||||
* All but the last two are decided before anything is built. `capacity-reached` and
|
||||
* `rebuild-failed` can only happen once a click is spending the plan, and they
|
||||
* are the two the banner must not confuse with a missing workspace: one means
|
||||
* "try again after closing something", the other means the CLI would not start.
|
||||
@@ -120,6 +124,7 @@ export interface RebootRestoreRejection {
|
||||
| 'no-persisted-record'
|
||||
| 'intentionally-ended'
|
||||
| 'not-running'
|
||||
| 'agent-exited'
|
||||
| 'respawn-blocked'
|
||||
| 'remote-or-docker'
|
||||
| 'unsupported-mode'
|
||||
@@ -191,7 +196,8 @@ export function planRebootRestore(
|
||||
//
|
||||
// ⚠️ This does NOT catch a session the user ended with `/exit`. See the
|
||||
// module header: that leaves the pid in place, because the pid is the tmux
|
||||
// attach process and `remain-on-exit` keeps it alive.
|
||||
// attach process and `remain-on-exit` keeps it alive. The `paneExit` check
|
||||
// below catches it instead.
|
||||
//
|
||||
// Conservative on purpose. A session that somehow persisted no pid while
|
||||
// genuinely running is not offered, and its conversation stays reachable
|
||||
@@ -200,6 +206,13 @@ export function planRebootRestore(
|
||||
skipped.push({ sessionId, reason: 'not-running' });
|
||||
continue;
|
||||
}
|
||||
if (isCleanPaneExit(state.paneExit)) {
|
||||
// The user ended the agent, and the clean-exit sweep would have closed the
|
||||
// session had the power not gone first (Ark0N/Codeman#446). The same
|
||||
// explicit-0 rule applies: an absent status is unknown, not clean.
|
||||
skipped.push({ sessionId, reason: 'agent-exited' });
|
||||
continue;
|
||||
}
|
||||
if (state.respawnBlocked === true) {
|
||||
// The crash-loop breaker tripped on this pane. Re-creating it restarts the loop.
|
||||
skipped.push({ sessionId, reason: 'respawn-blocked' });
|
||||
|
||||
@@ -540,6 +540,32 @@ export function remoteDisplayPath(
|
||||
return `${remote.username}@${remote.host}:${path}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Refresh HOST-level config on a RESTORED `SessionRemote`.
|
||||
*
|
||||
* A session's `remote` block is persisted at launch time (mux-sessions.json /
|
||||
* state.json) and recovery uses that snapshot, so a field ADDED to the host config
|
||||
* later never reaches an already-running session — not even across a Codeman
|
||||
* restart. That is exactly how a `wakeCommand` added to `remote-hosts.json` would
|
||||
* silently do nothing until the session is relaunched (which for an owned remote
|
||||
* session means killing the remote tmux).
|
||||
*
|
||||
* Deliberately narrow: ONLY `wakeCommand`/`wakeMac` are taken from the host config,
|
||||
* and the host is authoritative for them (removing one in the config turns that
|
||||
* wake path off again). The other host-level fields (`commands`, ssh options) stay as
|
||||
* persisted so this cannot silently change how an existing pane connects.
|
||||
*/
|
||||
export function rehydrateRemoteHostFields<T extends { hostId: string; wakeCommand?: string; wakeMac?: string }>(
|
||||
remote: T | undefined,
|
||||
hostsById: ReadonlyMap<string, RemoteHost>
|
||||
): T | undefined {
|
||||
if (!remote) return remote;
|
||||
const host = hostsById.get(remote.hostId);
|
||||
if (!host) return remote;
|
||||
if (remote.wakeCommand === host.wakeCommand && remote.wakeMac === host.wakeMac) return remote;
|
||||
return { ...remote, wakeCommand: host.wakeCommand, wakeMac: host.wakeMac };
|
||||
}
|
||||
|
||||
export function toSessionRemote(host: RemoteHost, remoteCase: RemoteCase): SessionRemote {
|
||||
return {
|
||||
hostId: host.id,
|
||||
@@ -549,6 +575,10 @@ export function toSessionRemote(host: RemoteHost, remoteCase: RemoteCase): Sessi
|
||||
port: host.port,
|
||||
remotePath: remoteCase.remotePath,
|
||||
commands: host.commands,
|
||||
// Wake-on-LAN command/MAC travel with the session so the input route can wake a
|
||||
// sleeping host without a second config read (see remote-wake.ts).
|
||||
wakeCommand: host.wakeCommand,
|
||||
wakeMac: host.wakeMac,
|
||||
// COD-105 — the COD-104 launch path creates the remote session, so we own it
|
||||
// (an explicit kill may propagate a remote kill-session). Discovered+attached
|
||||
// sessions go through `toAttachedSessionRemote` with `owned: false`.
|
||||
@@ -587,6 +617,10 @@ export function toAttachedSessionRemote(
|
||||
port: host.port,
|
||||
remotePath,
|
||||
commands: host.commands,
|
||||
// An attached session can be woken exactly the same way — the identity of the
|
||||
// creator does not change whether the host is asleep.
|
||||
wakeCommand: host.wakeCommand,
|
||||
wakeMac: host.wakeMac,
|
||||
// Discovered + attached — another Codeman created it. Detach-not-kill.
|
||||
owned: false,
|
||||
remoteSessionName,
|
||||
|
||||
+1035
File diff suppressed because it is too large
Load Diff
@@ -21,6 +21,8 @@
|
||||
* in 12/12 windows and the four idle ones in 0/12.
|
||||
*/
|
||||
|
||||
import { stripAnsi } from './utils/regex-patterns.js';
|
||||
|
||||
/**
|
||||
* A gap longer than this ends a run of continuous output. Claude repaints at
|
||||
* least once a second while working, so this leaves generous headroom.
|
||||
@@ -91,3 +93,124 @@ export function isSustainedActivity(streak: ActivityStreak | null, streakMs: num
|
||||
export function isPaneQuiet(lastActivityAt: number, now: number, silenceMs: number = IDLE_SILENCE_MS): boolean {
|
||||
return now - lastActivityAt >= silenceMs;
|
||||
}
|
||||
|
||||
/**
|
||||
* How many rows at the foot of a pane capture may hold the background-work row, for a
|
||||
* CLI that declares no number of its own (`capabilities.workDetect.watchingLines`).
|
||||
*
|
||||
* One, because the tightest window is the right default and Claude Code needs no more:
|
||||
* it draws its chip on the LAST row of the screen. Blank rows are dropped before the
|
||||
* window is taken, so a trailing blank costs nothing, and a CLI that ever prints a row
|
||||
* BELOW its chip loses the badge rather than gaining a hole.
|
||||
*
|
||||
* ⚠️ The size of this window is a trust boundary, not a tidiness measure, and the row
|
||||
* it excludes first is the one that taught us so: Claude's status line sits directly
|
||||
* above the footer, its content comes from a `statusLine` command, and a session running
|
||||
* with permissions bypassed can write that command into `.claude/settings.json` in its
|
||||
* own workspace. A window of two therefore let an agent print `· 1 monitor ·` onto a row
|
||||
* of its own and silence its own idle alert. Every row added here is another row
|
||||
* somebody may be able to write, so widen this only for a CLI whose layout forces it,
|
||||
* and never to a whole-pane search.
|
||||
*/
|
||||
export const WATCHING_TAIL_LINES = 1;
|
||||
|
||||
/** Longest label a badge will carry. A footer chip is a handful of words. */
|
||||
export const MAX_WATCHING_LABEL_CHARS = 40;
|
||||
|
||||
/**
|
||||
* What a pane says is still running in the background, e.g. `1 monitor` or `2 shells`.
|
||||
*
|
||||
* The CLI writes that chip while a monitor, a backgrounded shell or a cloud session it
|
||||
* started is still going, which is exactly the case where the agent has ended its turn
|
||||
* without wanting anything from the user. `pattern` comes from the CLI's own registry
|
||||
* entry (`capabilities.workDetect.watchingLine`); group 1 is the label when the pattern
|
||||
* declares one, and the whole match stands in when it does not.
|
||||
*
|
||||
* Each candidate row is tested on its own, bottom row first, so a pattern can anchor
|
||||
* itself with `^` or `$` against a single row rather than against a joined block. Blank
|
||||
* rows are dropped before the window is taken, because a CLI that leaves a blank line
|
||||
* between its chrome rows would otherwise spend the window on nothing. The answer is
|
||||
* stripped of ANSI and capped, because it ends up on a badge and in an approval card.
|
||||
*
|
||||
* @param tailLines how many non-blank rows from the bottom to look at, defaulting to
|
||||
* `WATCHING_TAIL_LINES`; a CLI declares its own when its row is not the last one
|
||||
* @returns the label, or null when the pane shows no background work
|
||||
*/
|
||||
export function watchingLabel(
|
||||
paneText: string | null | undefined,
|
||||
pattern: RegExp,
|
||||
tailLines: number = WATCHING_TAIL_LINES
|
||||
): string | null {
|
||||
if (!paneText) return null;
|
||||
const lines = stripAnsi(paneText)
|
||||
.split('\n')
|
||||
.map((line) => line.trimEnd())
|
||||
.filter((line) => line !== '');
|
||||
for (const line of lines.slice(-Math.max(1, tailLines)).reverse()) {
|
||||
// A pattern compiled by compileVersionRegex() never carries the `g` flag, but a
|
||||
// caller reaching in from a test or a config reload might, and a stale lastIndex
|
||||
// would make the same screen match every other call.
|
||||
pattern.lastIndex = 0;
|
||||
const match = pattern.exec(line);
|
||||
if (!match) continue;
|
||||
const label = (match[1] ?? match[0]).trim().slice(0, MAX_WATCHING_LABEL_CHARS);
|
||||
if (label) return label;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* How many rows above the composer the turn's closing row may sit. Between the two Claude
|
||||
* draws only its composer border and, sometimes, a right-aligned hint
|
||||
* (`new task? /clear to save 169.1k tokens`), so this leaves room for a blank row or two
|
||||
* and no more. A bound, not a tuning knob: the walk must never reach far enough up the
|
||||
* transcript to find an old turn's row.
|
||||
*/
|
||||
export const AWAITING_SEARCH_ROWS = 6;
|
||||
|
||||
/** A row that opens with a box-drawing character is the composer's frame, not transcript. */
|
||||
const COMPOSER_FRAME_ROW = /^[─-╿]/;
|
||||
|
||||
/**
|
||||
* Whether the pane's newest turn ended by handing off to workers the CLI will wait for,
|
||||
* e.g. Claude's `✻ Waiting for 1 dynamic workflow to finish`.
|
||||
*
|
||||
* Such a pane is quiet and shows its composer, so every other signal calls it idle, yet
|
||||
* nothing is being asked of the user: the CLI resumes by itself when the workers report
|
||||
* back. That is why a session in this state counts as working.
|
||||
*
|
||||
* ⚠️ The row is a snapshot. Claude renders it once, at the end of the turn, and never
|
||||
* updates it, so after the workers finish the same words are still on screen above the
|
||||
* follow-up turn. Matching them anywhere on the pane would pin the session busy for as
|
||||
* long as they stay visible. Only the newest transcript row counts: the walk starts at
|
||||
* the composer (the LAST row carrying `promptGlyph`), steps up past blank rows, the
|
||||
* composer's frame and anything indented (a right-aligned hint, a wrapped continuation),
|
||||
* and tests the first row that starts in column 0. A follow-up turn always puts rows of
|
||||
* its own there, so the stale copy is never the one tested.
|
||||
*
|
||||
* @param promptGlyph the CLI's composer glyph (`capabilities.workDetect.promptGlyph`)
|
||||
* @returns false when the screen shows no composer, which is no evidence either way
|
||||
*/
|
||||
export function isAwaitingWorkers(paneText: string | null | undefined, pattern: RegExp, promptGlyph: string): boolean {
|
||||
if (!paneText) return false;
|
||||
const rows = stripAnsi(paneText)
|
||||
.split('\n')
|
||||
.map((row) => row.trimEnd());
|
||||
let composer = -1;
|
||||
for (let i = rows.length - 1; i >= 0; i--) {
|
||||
// Claude has drawn its composer both bare (`❯ …` between rules) and boxed (`│ ❯ … │`).
|
||||
if (rows[i].replace(/^[\s│]+/, '').startsWith(promptGlyph)) {
|
||||
composer = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (composer < 0) return false;
|
||||
for (let i = composer - 1; i >= Math.max(0, composer - AWAITING_SEARCH_ROWS); i--) {
|
||||
const row = rows[i];
|
||||
if (row === '' || /^\s/.test(row) || COMPOSER_FRAME_ROW.test(row)) continue;
|
||||
// Same reasoning as watchingLabel(): a caller's `g` flag must not make this flap.
|
||||
pattern.lastIndex = 0;
|
||||
return pattern.test(row);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -6,13 +6,18 @@
|
||||
* stripped before the session is built. The create and resume routes are what
|
||||
* this bites on: they clamp what a request asked for.
|
||||
*
|
||||
* The reboot-restore route calls it as defence in depth, and today it can strip
|
||||
* nothing. `Session.getEnvOverridesForPersist()` keeps only `CLAUDE_CODE_*` and
|
||||
* `CLAUDE_CONFIG_DIR` out of a session's overrides, claude's `privilegedEnvKeys`
|
||||
* are the five `ANTHROPIC_*` names, and that pass admits claude alone — so a
|
||||
* persisted record cannot carry a clamped key. The call is there for the day the
|
||||
* persisted set widens. The grant re-resolution that does bite on that path is
|
||||
* `resolveClaudeModeForUsername`, which recomputes the permission mode.
|
||||
* The reboot-restore route calls it as defence in depth, and it CAN strip
|
||||
* something today: `Session.getEnvOverridesForPersist()` keeps only
|
||||
* `CLAUDE_CODE_*` and `CLAUDE_CONFIG_DIR` out of a session's overrides, and
|
||||
* claude's `privilegedEnvKeys` now includes both `CLAUDE_CODE_MAX_CONTEXT_TOKENS`
|
||||
* and `CLAUDE_CONFIG_DIR` (Custom Model Endpoint Profiles, since both can
|
||||
* redirect a claude session's traffic — see stock.ts's own comment on why they
|
||||
* are listed despite not needing the clamp for that feature). So a non-granted
|
||||
* owner's persisted `CLAUDE_CONFIG_DIR` (the per-client-account override, #255)
|
||||
* is now stripped on reboot-restore, silently returning that session to the
|
||||
* default Claude account rather than the account it was pointed at. The grant
|
||||
* re-resolution that ALSO bites on that path is `resolveClaudeModeForUsername`,
|
||||
* which recomputes the permission mode.
|
||||
*
|
||||
* This lives outside `web/routes` on purpose. The question it answers is about
|
||||
* session privilege rather than about HTTP, and `cron/cron-service.ts` sets the
|
||||
|
||||
@@ -0,0 +1,133 @@
|
||||
/**
|
||||
* @fileoverview Verify that a programmatically sent prompt actually LEFT the composer,
|
||||
* and press Enter again while it has not.
|
||||
*
|
||||
* Claude Code 2.1.277 (auto-installed 2026-09-18) takes typed text the moment its
|
||||
* composer paints but ignores Enter for the first 30 to 50 seconds after it, so the
|
||||
* `send-keys -l <text>` + `send-keys Enter` pair `TmuxManager.sendInput()` sends 50 ms
|
||||
* apart leaves the prompt sitting on the composer with `0 tokens`, and every caller
|
||||
* that then waits for the turn (send-and-wait, the agent skill, the maintainer bot,
|
||||
* cron, Ralph) burns its whole timeout on a turn that never started. Measured through
|
||||
* the input route on 2026-09-19: an Enter at 28 s stranded, one at 51 s submitted.
|
||||
*
|
||||
* The rule: after a write that carried a carriage return, read the pane on a short
|
||||
* schedule; while the LAST composer line (the CLI's own prompt glyph) still holds the
|
||||
* head of what was sent, send Enter again. An empty composer ends it, and so does a
|
||||
* composer holding anything else, because that text is the user's or the CLI's, never
|
||||
* ours. A pane with no composer line at all (a shell, a CLI whose glyph is not
|
||||
* declared, a direct-PTY session with no pane to read) does nothing: this runs for
|
||||
* EVERY programmatic sender, so a blind Enter here could confirm a dialog nobody asked
|
||||
* about. The composer is the last glyph line on purpose: Claude Code echoes a submitted
|
||||
* prompt with the same glyph higher up in the transcript, so only the last one says
|
||||
* whether the text was taken.
|
||||
*
|
||||
* Pure apart from the injected capture, send and log, so the schedule, the cap and
|
||||
* every stop condition are unit-tested with fake timers (test/session-submit-verifier.test.ts).
|
||||
*/
|
||||
import { stripAnsi } from './utils/index.js';
|
||||
|
||||
/**
|
||||
* When to look, counted from the write: 2 s catches the common case (taken) with one
|
||||
* capture, and the tail reaches 60 s, past twice the longest window measured. Enter is
|
||||
* re-sent at every check that still finds the prompt, so a 50 s window costs about
|
||||
* seven Enters and one capture each; a taken prompt costs one capture.
|
||||
*/
|
||||
export const SUBMIT_VERIFY_DELAYS_MS: readonly number[] = [
|
||||
2_000, 3_000, 5_000, 5_000, 5_000, 10_000, 10_000, 10_000, 10_000,
|
||||
];
|
||||
|
||||
/** How many leading characters of the prompt have to match, whitespace removed. */
|
||||
const PROMPT_HEAD_CHARS = 24;
|
||||
|
||||
const compact = (s: string): string => s.replace(/\s+/g, '');
|
||||
|
||||
/**
|
||||
* Whether `prompt` is still sitting unsubmitted in the composer of `screen`.
|
||||
*
|
||||
* - `true`: the last `glyph` line holds the prompt's head.
|
||||
* - `false`: the composer is empty (the prompt was taken) or holds other text.
|
||||
* - `undefined`: no composer line at all; nothing can be said, so nothing is sent.
|
||||
*
|
||||
* Whitespace is removed on both sides before comparing, because the composer wraps a
|
||||
* long prompt onto indented continuation lines and Claude Code draws a no-break space
|
||||
* after the glyph; `\s` covers that one in JavaScript.
|
||||
*/
|
||||
export function promptStillInComposer(screen: string, prompt: string, glyph: string): boolean | undefined {
|
||||
if (!glyph) return undefined;
|
||||
const composerLines = stripAnsi(screen)
|
||||
.split('\n')
|
||||
.map((l) => l.trim())
|
||||
.filter((l) => l.startsWith(glyph));
|
||||
if (composerLines.length === 0) return undefined;
|
||||
const composer = compact(composerLines[composerLines.length - 1].slice(glyph.length));
|
||||
if (!composer) return false;
|
||||
const head = compact(prompt).slice(0, PROMPT_HEAD_CHARS);
|
||||
return head.length > 0 && composer.startsWith(head);
|
||||
}
|
||||
|
||||
export interface SubmitVerifierDeps {
|
||||
/** The rendered pane, or null when there is none to read. */
|
||||
capture: () => string | null | undefined;
|
||||
/** Press Enter once. Failures are swallowed; the next check decides again. */
|
||||
sendEnter: () => Promise<unknown> | unknown;
|
||||
/** The CLI's composer glyph, resolved at check time (the registry can change). */
|
||||
glyph: () => string;
|
||||
log?: (message: string) => void;
|
||||
/** Test seam; production uses SUBMIT_VERIFY_DELAYS_MS. */
|
||||
delaysMs?: readonly number[];
|
||||
}
|
||||
|
||||
/**
|
||||
* One per session. `arm(text)` starts the schedule for the prompt just sent and
|
||||
* cancels any earlier one: a newer write owns the composer now, and re-sending Enter
|
||||
* for an older prompt could submit the newer one early. `cancel()` is for teardown.
|
||||
*/
|
||||
export class SubmitVerifier {
|
||||
private timer: NodeJS.Timeout | null = null;
|
||||
private generation = 0;
|
||||
|
||||
constructor(private readonly deps: SubmitVerifierDeps) {}
|
||||
|
||||
arm(text: string): void {
|
||||
this.cancel();
|
||||
const gen = this.generation;
|
||||
const delays = this.deps.delaysMs ?? SUBMIT_VERIFY_DELAYS_MS;
|
||||
let step = 0;
|
||||
let elapsed = 0;
|
||||
let resent = 0;
|
||||
|
||||
const schedule = (): void => {
|
||||
if (step >= delays.length) return;
|
||||
const delay = delays[step++];
|
||||
elapsed += delay;
|
||||
this.timer = setTimeout(() => void check(), delay);
|
||||
this.timer.unref?.();
|
||||
};
|
||||
const check = async (): Promise<void> => {
|
||||
this.timer = null;
|
||||
if (gen !== this.generation) return;
|
||||
const screen = this.deps.capture();
|
||||
if (promptStillInComposer(screen ?? '', text, this.deps.glyph()) !== true) return;
|
||||
resent++;
|
||||
this.deps.log?.(
|
||||
`prompt still in the composer after ${Math.round(elapsed / 1000)}s, re-sending Enter (${resent}/${delays.length})`
|
||||
);
|
||||
try {
|
||||
await this.deps.sendEnter();
|
||||
} catch {
|
||||
// The next check re-reads the screen and decides again.
|
||||
}
|
||||
if (gen !== this.generation) return;
|
||||
schedule();
|
||||
};
|
||||
schedule();
|
||||
}
|
||||
|
||||
cancel(): void {
|
||||
this.generation++;
|
||||
if (this.timer) {
|
||||
clearTimeout(this.timer);
|
||||
this.timer = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
+583
-25
@@ -60,8 +60,11 @@ import {
|
||||
type SessionDocker,
|
||||
type SessionNameSource,
|
||||
type SessionWriteOptions,
|
||||
type PaneExit,
|
||||
} from './types.js';
|
||||
import { resolveAndClaimOmpSessionId } from './utils/omp-session-resolver.js';
|
||||
import { claudeTranscriptExists } from './utils/claude-transcript.js';
|
||||
import { matchesPattern } from './config/cli-registry/patterns.js';
|
||||
import { probeDockerCliVersion } from './docker-hosts.js';
|
||||
import { probeRemoteCliVersion } from './remote-hosts.js';
|
||||
import type { TerminalMultiplexer, MuxSession } from './mux-interface.js';
|
||||
@@ -81,6 +84,9 @@ import {
|
||||
trackActivityStreak,
|
||||
isSustainedActivity,
|
||||
isPaneQuiet,
|
||||
watchingLabel,
|
||||
isAwaitingWorkers,
|
||||
WATCHING_TAIL_LINES,
|
||||
IDLE_RECHECK_MS,
|
||||
PANE_PROBE_MIN_INTERVAL_MS,
|
||||
PANE_PROBE_RECHECK_MS,
|
||||
@@ -110,6 +116,7 @@ import {
|
||||
import { DEFAULT_TMUX_HISTORY_LIMIT } from './config/terminal-history.js';
|
||||
import { EXEC_TIMEOUT_MS } from './config/exec-timeout.js';
|
||||
import { getCli } from './config/cli-registry/registry.js';
|
||||
import { SubmitVerifier } from './session-submit-verifier.js';
|
||||
import { compileVersionRegex } from './config/cli-registry/patterns.js';
|
||||
import { resolveSessionCliVersion } from './utils/cli-resolver.js';
|
||||
import {
|
||||
@@ -496,12 +503,43 @@ export class Session extends EventEmitter {
|
||||
private _activityStreak: ActivityStreak | null = null; // Unbroken run of PTY repaints (working detection)
|
||||
private _lastPaneProbeAt = 0; // Throttle for the tmux screen probe
|
||||
private _lastPaneProbeWorking: boolean | null = null; // Its last verdict (null = could not read)
|
||||
/**
|
||||
* Background work the pane's own footer reports, e.g. `1 monitor`; null for none.
|
||||
*
|
||||
* Cached BESIDE `_lastPaneProbeWorking` and refreshed only by a capture that really
|
||||
* happened, so it goes stale exactly as that verdict does. The probe returns its
|
||||
* cached boolean without re-capturing inside `PANE_PROBE_MIN_INTERVAL_MS`, and a
|
||||
* label derived from a capture nobody took would be a guess wearing a fact's clothes.
|
||||
*
|
||||
* ⚠️ It then FREEZES once `_confirmIdle()` concludes: `activityTimeout` is null from
|
||||
* there, and nothing looks at the pane again until it produces output. That is
|
||||
* correct rather than merely tolerable, because work ending repaints the pane either
|
||||
* way — a monitor firing wakes the agent, and codex drops its background-terminal row
|
||||
* on its own. Do not add a timer to keep this fresh; it would spend a `capture-pane`
|
||||
* per idle session per tick to learn nothing.
|
||||
*
|
||||
* A server restart is not a hole in that either, though it looks like one: this field
|
||||
* is live state and starts empty. Reconciliation re-attaches the pane, the attach
|
||||
* repaint carries the composer glyph, and the idle confirmation that arms on it probes
|
||||
* and re-reads the label with no input from anyone — measured 2026-09-23 on a restarted
|
||||
* instance, back within ~20 s for a session whose background terminal was still
|
||||
* running. A session that comes back with no label has no chip on its screen.
|
||||
*/
|
||||
private _watching: string | null = null;
|
||||
/** Lazily compiled `capabilities.workDetect.workingLine`. See _workingLinePattern(). */
|
||||
private _workingLineRe: RegExp | undefined = undefined;
|
||||
/** Lazily compiled `capabilities.workDetect.watchingLine`. See _watchingLinePattern(). */
|
||||
private _watchingLineRe: RegExp | null | undefined = undefined;
|
||||
/** Resolved with the pattern above: how many rows at the foot of the screen to search. */
|
||||
private _watchingWindow = WATCHING_TAIL_LINES;
|
||||
/** Lazily compiled `capabilities.workDetect.awaitingLine`. See _awaitingLinePattern(). */
|
||||
private _awaitingLineRe: RegExp | null | undefined = undefined;
|
||||
private _trustDialogAccepted: boolean = false; // Stops the trust-dialog scan (answered, or given up)
|
||||
private _trustDialogAttempts = 0; // Keystrokes sent at the trust dialog
|
||||
private _lastTrustDialogScanAt = 0; // Throttle for the trust-dialog screen read
|
||||
private _trustDialogTimer: NodeJS.Timeout | null = null; // Re-read after a keystroke (see below)
|
||||
/** Re-sends Enter while a programmatic prompt still sits in the composer (session-submit-verifier.ts). */
|
||||
private _submitVerifier: SubmitVerifier | null = null;
|
||||
private _interactiveStartedAt = 0; // When the interactive pane launched (bounds that scan)
|
||||
private _taskTracker: TaskTracker;
|
||||
|
||||
@@ -539,6 +577,35 @@ export class Session extends EventEmitter {
|
||||
private _mux: TerminalMultiplexer | null = null;
|
||||
private _muxSession: MuxSession | null = null;
|
||||
private _useMux: boolean = false;
|
||||
/**
|
||||
* The agent in this session's local tmux pane has exited (Ark0N/Codeman#446).
|
||||
* `null` is the UNKNOWN arm of the tri-state and is what {@link setPaneExit}
|
||||
* stores for every session shape the field does not apply to. See
|
||||
* {@link PaneExit} for the shapes and for why an unknown answer must never be
|
||||
* rendered as "alive".
|
||||
*/
|
||||
private _paneExit: PaneExit | null = null;
|
||||
/**
|
||||
* How many starts, attaches or relaunches are running for this session's
|
||||
* pane. While one is, a dead-pane reading may describe a pane that is being
|
||||
* revived on purpose, so the exited-agent sweep leaves the session alone
|
||||
* (Ark0N/Codeman#446). A counter rather than a flag, so two overlapping
|
||||
* operations cannot clear each other's mark.
|
||||
*/
|
||||
private _paneLifecycleOps = 0;
|
||||
/** When the last pane start, attach or relaunch finished (ms), 0 when none has run. */
|
||||
private _paneStartedAt = 0;
|
||||
/**
|
||||
* The server has started closing this session, so no start or attach may
|
||||
* begin (see {@link markClosing}).
|
||||
*/
|
||||
private _closing = false;
|
||||
/**
|
||||
* This session was rebuilt from the tmux socket rather than from Codeman's
|
||||
* own records, so its `remote`/`docker` metadata is missing rather than known
|
||||
* to be absent. See {@link MuxSession.discovered}.
|
||||
*/
|
||||
private _discoveredMuxSession = false;
|
||||
// Flag to prevent new timers after session is stopped
|
||||
private _isStopped: boolean = false;
|
||||
|
||||
@@ -715,6 +782,10 @@ export class Session extends EventEmitter {
|
||||
lastSubmitAt?: number;
|
||||
/** Restored conversation chain, oldest first (see `claudeSessionChain`). */
|
||||
claudeSessionChain?: string[];
|
||||
/** Restored agent-exit observation for this session's pane (see `paneExit`). */
|
||||
paneExit?: PaneExit;
|
||||
/** This session was rebuilt from the tmux socket, so its metadata is a guess. */
|
||||
discoveredMuxSession?: boolean;
|
||||
/** Restored wall-clock ms of the pane's last output (recovery only; see `_wireActivityAt`). */
|
||||
lastActivityAt?: number;
|
||||
/** Remote execution metadata for sessions launched through SSH inside local tmux. */
|
||||
@@ -867,6 +938,16 @@ export class Session extends EventEmitter {
|
||||
this._remote = config.remote;
|
||||
this._docker = config.docker;
|
||||
this._owner = config.owner;
|
||||
this._discoveredMuxSession = config.discoveredMuxSession === true;
|
||||
// Restored so a record that says the agent exited survives a server restart
|
||||
// rather than being blanked by the first persist after boot. It runs here
|
||||
// because the scoping reads `_remote`, `_docker` and the mux fields, all of
|
||||
// which are set by now. It is a claim about a pane this process has not
|
||||
// looked at yet, so every path that starts or re-attaches a pane drops it
|
||||
// (see `_setupOrAttachMuxSession`) and the pane-exit watcher's own tick
|
||||
// replaces it with a first-hand reading. NOT the stats collector, which a
|
||||
// browser panel arms and disarms — see `startPaneExitWatcher`.
|
||||
this.setPaneExit(config.paneExit);
|
||||
// Never self-parent: a session pointing at itself would draw a zero-length
|
||||
// lineage arc under its own tab. Only reachable via the recovery path, where
|
||||
// both the id and the saved parent come from disk.
|
||||
@@ -1094,6 +1175,118 @@ export class Session extends EventEmitter {
|
||||
return this._muxSession?.muxName ?? null;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when a tmux pane's death would mean THIS session's agent has exited.
|
||||
*
|
||||
* Four shapes fail the test, and each would otherwise publish a death that is
|
||||
* not the agent's. A direct-PTY session owns no pane at all. A remote SSH
|
||||
* session's local pane holds the ssh client, whose death means a transport
|
||||
* drop OR an exit, which is the ambiguity PR #355 was about. A docker case's
|
||||
* local pane holds a `docker exec` into the container's own tmux.
|
||||
*
|
||||
* The fourth is a session rebuilt from the socket. Absent `remote`/`docker`
|
||||
* normally means "this is local", but on a discovered record it only means
|
||||
* "Codeman never found the metadata": the synthetic `restored-<fragment>` id
|
||||
* matches no `state.json` entry, so a remote session rediscovered after
|
||||
* `mux-sessions.json` was lost arrives looking local, and its next transport
|
||||
* drop would be published as an agent exit. Unproven locality fails closed.
|
||||
*/
|
||||
private get paneExitApplies(): boolean {
|
||||
if (this._discoveredMuxSession) return false;
|
||||
return this._useMux && this._muxSession !== null && !this._remote && !this._docker;
|
||||
}
|
||||
|
||||
/** What Codeman last observed of this pane's agent, or undefined for UNKNOWN. */
|
||||
get paneExit(): PaneExit | undefined {
|
||||
return this._paneExit ?? undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* True while a start, attach or relaunch is running for this session's pane.
|
||||
* The exited-agent sweep reads it (see `pane-exit-sweep.ts`): the dead-pane
|
||||
* branch of {@link _setupOrAttachMuxSession} respawns an exited pane, and
|
||||
* until it finishes and clears the exit, the pane still reads as dead.
|
||||
*/
|
||||
get paneLifecycleInFlight(): boolean {
|
||||
return this._paneLifecycleOps > 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* When the last start, attach or relaunch of this pane finished, or 0 when
|
||||
* none has run in this process. The exited-agent sweep keeps an exit that
|
||||
* lands within `CLEAN_EXIT_MIN_PANE_LIFETIME_MS` of it, since that reads as a
|
||||
* CLI failing at startup rather than a user ending it. An attach to a pane
|
||||
* that was already running stamps it too, which only costs a user who
|
||||
* `/exit`s within seconds of a server restart a row to close by hand.
|
||||
*/
|
||||
get paneStartedAt(): number {
|
||||
return this._paneStartedAt;
|
||||
}
|
||||
|
||||
/**
|
||||
* Mark this session as being closed, or clear the mark after a close that
|
||||
* failed. While it is set, {@link startInteractive} and {@link startShell}
|
||||
* refuse to run. A start that raced a close would otherwise launch a CLI in a
|
||||
* tmux session whose record is about to be deleted (Ark0N/Codeman#446).
|
||||
*/
|
||||
markClosing(closing: boolean): void {
|
||||
this._closing = closing;
|
||||
}
|
||||
|
||||
/** Run one pane start, attach or relaunch with {@link paneLifecycleInFlight} raised. */
|
||||
private async _withPaneLifecycle<T>(op: () => Promise<T>): Promise<T> {
|
||||
this._paneLifecycleOps++;
|
||||
try {
|
||||
return await op();
|
||||
} finally {
|
||||
this._paneLifecycleOps--;
|
||||
this._paneStartedAt = Date.now();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Forget this pane's exit, on both this record and the mux layer's cache.
|
||||
* Every path that starts or relaunches a command in the pane calls it, and
|
||||
* the mux half also invalidates a pane read already in flight.
|
||||
*
|
||||
* It does not persist or broadcast by itself; the caller owns both. ⚠ That
|
||||
* caller MUST persist, and the pane-exit watcher is not a fallback for it:
|
||||
* the watcher's next tick reads UNKNOWN, finds this field already cleared,
|
||||
* reports no change and therefore writes nothing, so a caller that only
|
||||
* broadcasts leaves `state.json` saying the agent exited for as long as the
|
||||
* session stays quiet. `/interactive` and `/shell` did exactly that until
|
||||
* Ark0N/Codeman#446 review; both now persist on their success path.
|
||||
*/
|
||||
private clearPaneExitForNewPane(): void {
|
||||
this.setPaneExit(undefined);
|
||||
if (this._muxSession) this._mux?.clearPaneExit?.(this._muxSession.muxName);
|
||||
}
|
||||
|
||||
/**
|
||||
* Record what the mux layer observed of this pane's agent, and say whether
|
||||
* that changed the answer. The caller persists and broadcasts on a true.
|
||||
*
|
||||
* A session the field does not apply to is forced to UNKNOWN here rather than
|
||||
* at the reporting end, so the rule lives in one place and the mux layer stays
|
||||
* free to report the raw pane reading its own remote-reconnect watcher needs.
|
||||
*/
|
||||
setPaneExit(next: PaneExit | undefined): boolean {
|
||||
const resolved = this.paneExitApplies ? (next ?? null) : null;
|
||||
const prev = this._paneExit;
|
||||
if (prev === resolved) return false;
|
||||
if (
|
||||
prev !== null &&
|
||||
resolved !== null &&
|
||||
prev.status === resolved.status &&
|
||||
prev.signal === resolved.signal &&
|
||||
prev.at === resolved.at
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
this._paneExit = resolved;
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when this session's PTY is a tmux client rather than the program itself.
|
||||
* Read by the replay-side alt-screen strip, which must apply the same
|
||||
@@ -1115,6 +1308,16 @@ export class Session extends EventEmitter {
|
||||
return this._isWorking;
|
||||
}
|
||||
|
||||
/**
|
||||
* What the pane says is still running in the background, e.g. `1 monitor`, or null when
|
||||
* nothing is. A session with a label here has ended its turn without wanting anything
|
||||
* from the user, so a surface that would otherwise file it under "needs you" can say
|
||||
* what it is waiting for instead.
|
||||
*/
|
||||
get watching(): string | null {
|
||||
return this._watching;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if the session's process tree has active child processes beyond Claude itself.
|
||||
* Detects running bash tools, test suites, builds, servers, etc. that Claude spawned.
|
||||
@@ -1418,6 +1621,19 @@ export class Session extends EventEmitter {
|
||||
return this._nameSource;
|
||||
}
|
||||
|
||||
/**
|
||||
* The name to pin on the Claude CLI as `--name`, or undefined to let Claude
|
||||
* title the conversation itself. `--name` is the prompt-box label, the
|
||||
* `/resume` picker entry and the terminal title all at once, and a pinned
|
||||
* title stops Claude generating its own, so only a name the user chose is
|
||||
* worth pinning. Pinning the `w1-myapp` placeholder gave every conversation
|
||||
* in a case the same `/resume` entry; an auto name is a cut of the first
|
||||
* prompt, which Claude's own generated title already beats.
|
||||
*/
|
||||
get cliPinnedName(): string | undefined {
|
||||
return this._nameSource === 'manual' ? this._name : undefined;
|
||||
}
|
||||
|
||||
setAutoClear(enabled: boolean, threshold?: number): void {
|
||||
this._autoOps.setAutoClear(enabled, threshold);
|
||||
}
|
||||
@@ -1617,6 +1833,10 @@ export class Session extends EventEmitter {
|
||||
// by the constructor: a Codeman restart starts with a fresh breaker so boot
|
||||
// recovery can re-attach.
|
||||
respawnBlocked: this._respawnBlocked || undefined,
|
||||
// Ark0N/Codeman#446 — the agent in this pane has exited, published here so
|
||||
// it rides the existing `session:updated` broadcast and lands in state.json
|
||||
// through the same persist. `status` and `pid` above stay untouched by it.
|
||||
paneExit: this._paneExit ?? undefined,
|
||||
attachmentHistory: this.attachmentHistory.length > 0 ? this.attachmentHistory : undefined,
|
||||
lastSubmitAt: this._lastSubmitAt || undefined,
|
||||
// Only a chain the CLI's own hooks vouched for is persisted, and only when
|
||||
@@ -1672,6 +1892,7 @@ export class Session extends EventEmitter {
|
||||
totalCost: this._totalCost,
|
||||
messageCount: this._messages.length,
|
||||
isWorking: this._isWorking,
|
||||
watching: this._watching,
|
||||
lastPromptTime: this._lastPromptTime,
|
||||
// Buffer statistics for monitoring long-running sessions
|
||||
bufferStats: {
|
||||
@@ -1728,41 +1949,90 @@ export class Session extends EventEmitter {
|
||||
respawnPaneOptions: import('./mux-interface.js').RespawnPaneOptions;
|
||||
createSessionOptions: import('./mux-interface.js').CreateSessionOptions;
|
||||
spawnErrLabel: string;
|
||||
}): Promise<{ isRestored: boolean }> {
|
||||
}): Promise<{ isRestored: boolean; respawnedResumeId?: string; respawnedDeadPane: boolean }> {
|
||||
return this._withPaneLifecycle(() => this._doSetupOrAttachMuxSession(options));
|
||||
}
|
||||
|
||||
private async _doSetupOrAttachMuxSession(options: {
|
||||
respawnPaneOptions: import('./mux-interface.js').RespawnPaneOptions;
|
||||
createSessionOptions: import('./mux-interface.js').CreateSessionOptions;
|
||||
spawnErrLabel: string;
|
||||
}): Promise<{ isRestored: boolean; respawnedResumeId?: string; respawnedDeadPane: boolean }> {
|
||||
const mux = this._mux!;
|
||||
|
||||
// Verify stale mux session — tmux may have been destroyed (e.g., killed externally)
|
||||
// Verify stale mux session — tmux may have been destroyed (e.g., killed externally).
|
||||
// A session that HAD a mux session relaunches its CLI below just like a failed
|
||||
// respawn does (tmux kill-server, a tmux crash, an external kill-session), so
|
||||
// its transcript collides with the bare `--session-id` the same way. A
|
||||
// genuinely new session starts with `_muxSession` null and never sets this.
|
||||
let muxSessionVanished = false;
|
||||
if (this._muxSession && !mux.muxSessionExists(this._muxSession.muxName)) {
|
||||
console.log('[Session] Stale mux session detected (tmux gone):', this._muxSession.muxName);
|
||||
this._muxSession = null;
|
||||
muxSessionVanished = true;
|
||||
}
|
||||
|
||||
// Check if session exists but pane is dead (remain-on-exit keeps it alive)
|
||||
// Respawn the pane instead of creating a whole new session — preserves tmux scrollback
|
||||
let needsNewSession = false;
|
||||
let respawnedDeadPane = false;
|
||||
let respawnedResumeId: string | undefined;
|
||||
if (this._muxSession && mux.isPaneDead(this._muxSession.muxName)) {
|
||||
console.log('[Session] Dead pane detected, respawning:', this._muxSession.muxName);
|
||||
// Confirmed dead — safe to resolve/pin now (see `_pinOmpRespawnId()`).
|
||||
// `options.respawnPaneOptions` was built eagerly before this dead-pane
|
||||
// check ran, so it still carries the pre-pin ompConfig; rebuild it.
|
||||
this._pinOmpRespawnId();
|
||||
const newPid = await mux.respawnPane(this._buildRespawnPaneOptions());
|
||||
const respawnOptions = await this._buildRespawnPaneOptionsWithResumePin();
|
||||
const newPid = await mux.respawnPane(respawnOptions);
|
||||
if (!newPid) {
|
||||
console.error('[Session] Failed to respawn pane, will create new session');
|
||||
needsNewSession = true;
|
||||
} else {
|
||||
respawnedDeadPane = true;
|
||||
respawnedResumeId = respawnOptions.resumeSessionId;
|
||||
this._pendingEnvUnsets.clear();
|
||||
// Wait a moment for the respawned process to fully start
|
||||
await new Promise((resolve) => setTimeout(resolve, MUX_STARTUP_DELAY_MS));
|
||||
}
|
||||
}
|
||||
|
||||
// Whatever the last reading said about the OLD command in this pane is now
|
||||
// history: the branch above either respawned the pane or found it alive, and
|
||||
// the branch below creates a new one. The paths that reach here are boot
|
||||
// recovery and an explicit start, NOT a click on an exited tab — the browser
|
||||
// re-attaches only on a null pid, and the premise of Ark0N/Codeman#446 is
|
||||
// that an exited pane keeps its pid. `restartCli()` clears separately.
|
||||
this.clearPaneExitForNewPane();
|
||||
|
||||
// Check if we already have a mux session (restored session)
|
||||
const isRestored = this._muxSession !== null && !needsNewSession;
|
||||
if (isRestored) {
|
||||
console.log('[Session] Attaching to existing mux session:', this._muxSession!.muxName);
|
||||
} else {
|
||||
// Create a new mux session
|
||||
// Create a new mux session. When this is the FALLBACK after a failed
|
||||
// respawn, the eagerly-built create options still carry the unpinned
|
||||
// launch seed, so a session whose transcript exists would meet the same
|
||||
// `--session-id ... already in use` refusal the respawn just lost to —
|
||||
// the recovery of last resort failing for the very reason it was needed.
|
||||
// A genuinely new session has no transcript under any of its candidate
|
||||
// ids, so nothing is pinned and its command shape is unchanged.
|
||||
//
|
||||
// `_resumeSessionId` is written alongside, not just the create options:
|
||||
// this branch leaves `isRestored` false, so the block that sets
|
||||
// `_claudeSessionId` below reads that field and would otherwise settle on
|
||||
// `this.id` while the CLI resumes the chain tail. The response viewer,
|
||||
// Read My Mind and the unified-list alias map all read `_claudeSessionId`
|
||||
// until the next first-hand hook, so the two have to name the same
|
||||
// conversation. The vanished-tmux-session branch above relaunches for the
|
||||
// same reason and takes the same pin.
|
||||
if (needsNewSession || muxSessionVanished) {
|
||||
const pinned = (await this._buildRespawnPaneOptionsWithResumePin()).resumeSessionId;
|
||||
if (pinned) {
|
||||
options.createSessionOptions.resumeSessionId = pinned;
|
||||
this._resumeSessionId = pinned;
|
||||
}
|
||||
}
|
||||
this._muxSession = await mux.createSession(options.createSessionOptions);
|
||||
console.log('[Session] Created mux session:', this._muxSession.muxName);
|
||||
// No extra sleep — createSession() already waits for tmux readiness
|
||||
@@ -1797,13 +2067,14 @@ export class Session extends EventEmitter {
|
||||
env: buildMuxAttachEnv(cliExportsTruecolor(this.mode)),
|
||||
})
|
||||
);
|
||||
this._notePtySpawnGeometry(ptyCols, ptyRows);
|
||||
} catch (spawnErr) {
|
||||
console.error(`[Session] Failed to spawn PTY for ${options.spawnErrLabel}:`, spawnErr);
|
||||
this.emit('error', `Failed to attach to mux session: ${spawnErr}`);
|
||||
throw spawnErr;
|
||||
}
|
||||
|
||||
return { isRestored };
|
||||
return { isRestored, respawnedResumeId, respawnedDeadPane };
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1841,6 +2112,9 @@ export class Session extends EventEmitter {
|
||||
console.error('[Session] reattachRemote: respawnPane failed for', this._muxSession.muxName);
|
||||
return false;
|
||||
}
|
||||
// No-op for the record (a remote session's field is always UNKNOWN), but the
|
||||
// mux layer's cache is keyed by muxName and this pane now runs a new client.
|
||||
this.clearPaneExitForNewPane();
|
||||
console.log('[Session] reattachRemote: reattached remote session', this._muxSession.muxName, 'pid', newPid);
|
||||
return true;
|
||||
}
|
||||
@@ -1866,6 +2140,10 @@ export class Session extends EventEmitter {
|
||||
* the mux session is gone — see {@link reattachRemote} for that reasoning).
|
||||
*/
|
||||
async restartCli(): Promise<boolean> {
|
||||
return this._withPaneLifecycle(() => this._doRestartCli());
|
||||
}
|
||||
|
||||
private async _doRestartCli(): Promise<boolean> {
|
||||
if (!this._useMux || !this._mux || !this._muxSession) return false;
|
||||
const mux = this._mux;
|
||||
|
||||
@@ -1875,26 +2153,16 @@ export class Session extends EventEmitter {
|
||||
}
|
||||
|
||||
this._pinOmpRespawnId();
|
||||
const options = this._buildRespawnPaneOptions();
|
||||
// Unlike the dead-pane respawn, this one kills a WORKING pane whose conversation
|
||||
// already has a transcript, and a CLI that launches with `--session-id <id>` refuses
|
||||
// an id that is already in use (claude: `Error: Session ID ... is already in use.`),
|
||||
// which turned an endpoint switch into a dead pane and a lost session. A launch that
|
||||
// declares a `fallback` chain renders `resume || new` once a resume id is set, the
|
||||
// same `--resume <id> || --session-id <id>` shape the docker and remote pane commands
|
||||
// already use, so pin the live conversation id for THIS respawn only. The registry
|
||||
// shape is the gate, not the CLI's name: an entry whose resume id is minted by the
|
||||
// CLI itself (codex/pi/omp/grok) never declares that chain, and its resume field is
|
||||
// read from its own `<Mode>Config` rather than this top-level one anyway.
|
||||
if (!options.resumeSessionId && getCli(this.mode)?.launch.chain === 'fallback') {
|
||||
options.resumeSessionId = this._claudeSessionId ?? this.id;
|
||||
}
|
||||
const newPid = await mux.respawnPane(options);
|
||||
const newPid = await mux.respawnPane(await this._buildRespawnPaneOptionsWithResumePin());
|
||||
if (!newPid) {
|
||||
console.error('[Session] restartCli: respawnPane failed for', this._muxSession.muxName);
|
||||
return false;
|
||||
}
|
||||
this._pendingEnvUnsets.clear();
|
||||
// A relaunch in the same pane, so any exit observed of the previous command
|
||||
// is history. Without this the caller's persist-and-broadcast writes the old
|
||||
// exit straight back onto a session that is running again.
|
||||
this.clearPaneExitForNewPane();
|
||||
console.log('[Session] restartCli: restarted CLI for', this._muxSession.muxName, 'pid', newPid);
|
||||
return true;
|
||||
}
|
||||
@@ -1911,6 +2179,7 @@ export class Session extends EventEmitter {
|
||||
workingDir: this.workingDir,
|
||||
mode: this.mode,
|
||||
name: this._name,
|
||||
cliName: this.cliPinnedName,
|
||||
niceConfig: this._niceConfig,
|
||||
model: this._model,
|
||||
claudeMode: this._claudeMode,
|
||||
@@ -1944,6 +2213,112 @@ export class Session extends EventEmitter {
|
||||
return this._withCustomModelLaunchModel(options);
|
||||
}
|
||||
|
||||
/**
|
||||
* Respawn options for a pane whose command is being REPLACED, with the
|
||||
* conversation pinned so the relaunch resumes rather than collides.
|
||||
*
|
||||
* A CLI that launches with `--session-id <id>` refuses an id that is already
|
||||
* in use (claude: `Error: Session ID ... is already in use.`), and every
|
||||
* session whose agent has been prompted owns a transcript under that id. So
|
||||
* relaunching such a pane with the bare launch line fails, the pane dies
|
||||
* again immediately, and the user's conversation is stranded. A launch that
|
||||
* declares a `fallback` chain renders `resume || new` once a resume id is
|
||||
* set, which is the shape that survives both cases.
|
||||
*
|
||||
* Three candidates are tried in priority order — the conversation chain's
|
||||
* tail, the launch seed, then the session's own id — and the first one a
|
||||
* transcript backs is pinned. Four conditions gate that walk, each protecting
|
||||
* against a way of resuming the WRONG conversation or of making a working
|
||||
* relaunch fail.
|
||||
*
|
||||
* ⚠️ **A remote or docker session is never pinned.** Unlike `restartCli()`,
|
||||
* whose route refuses both, the dead-pane respawn is reached by every session
|
||||
* shape. Their pane commands (`buildRemoteLaunchCommand`,
|
||||
* `claudeDockerPaneCommand`) already render a SELF-HEALING
|
||||
* `--session-id <sid> || --resume <sid>`, and both flip to resume-first the
|
||||
* moment the resume id differs from the session id. The conversation lives on
|
||||
* the far side, so a local id pinned onto it resolves to nothing there, the
|
||||
* resume fails, and the `--session-id` fallback then collides with the
|
||||
* transcript the far side really does hold — both branches fail and the pane
|
||||
* dies. `_pinOmpRespawnId()` refuses remote for the same reason.
|
||||
*
|
||||
* ⚠️ **The candidates come from the conversation CHAIN, never from
|
||||
* `_claudeSessionId`.** That field holds either a first-hand id from the
|
||||
* CLI's own hook payload or a history correlation, which is a guess keyed on
|
||||
* the working directory. `_recordClaudeSessionInChain()` refuses a guess
|
||||
* precisely so it cannot "write a foreign conversation into this pane's
|
||||
* permanent record", and launching from one would do worse than the display
|
||||
* bug that rule exists to prevent: the relaunched CLI would open and WRITE to
|
||||
* a conversation that was never this pane's. The chain's tail is the live
|
||||
* conversation and is hook-vouched, so it leads the walk, ahead of the launch
|
||||
* seed, which is written once at construction and never moves off a `/clear`.
|
||||
*
|
||||
* ⚠️ **Every candidate must be backed by a transcript, the session's own id
|
||||
* included, and a candidate that has none is passed over rather than ending
|
||||
* the walk.** A pin that differs from the session id leaves
|
||||
* `--session-id <this.id>` in the fallback branch, so a resume that finds
|
||||
* nothing collides there and the pane dies exactly as it did before this
|
||||
* pinning existed. Pinning `this.id` renders the self-healing
|
||||
* `--resume <id> || --session-id <id>`, which is correct whether or not a
|
||||
* transcript exists, but a pane that has none pays for the shape twice:
|
||||
* claude prints "No conversation found" into the scrollback of a session that
|
||||
* is brand new, and `wrapWithNice()` prefixes only the FIRST branch of the
|
||||
* rendered `a || b`, so the branch that actually runs loses its priority for
|
||||
* the life of the session. Falling off the end of the walk therefore adds
|
||||
* no pin (the options keep any launch seed they already carried), which is
|
||||
* the right answer: with no transcript anywhere there is nothing for the
|
||||
* bare `--session-id <this.id>` to collide with.
|
||||
*
|
||||
* The create route pre-validates a resume id for the same reason, though it
|
||||
* additionally requires the transcript be substantial — here mere existence
|
||||
* is the question, because a one-line transcript still makes `--session-id`
|
||||
* collide.
|
||||
*
|
||||
* The registry shape is the last gate, not the CLI's name: an entry whose
|
||||
* resume id is minted by the CLI itself (codex/pi/omp/grok) declares no
|
||||
* `fallback` chain and reads its resume field from its own `<Mode>Config`.
|
||||
*
|
||||
* `reattachRemote()` deliberately does NOT call this. It re-runs the remote
|
||||
* session command, which attaches to the durable remote tmux with the agent
|
||||
* still inside it and renders no local `--session-id` to collide.
|
||||
*/
|
||||
private async _buildRespawnPaneOptionsWithResumePin(): Promise<import('./mux-interface.js').RespawnPaneOptions> {
|
||||
const options = this._buildRespawnPaneOptions();
|
||||
if (this._remote || this._docker) return options;
|
||||
const entry = getCli(this.mode);
|
||||
if (entry?.launch.chain !== 'fallback') return options;
|
||||
|
||||
const resumeIdPattern = entry.launch.params?.resumeId;
|
||||
const configDir = this._claudeConfigDir();
|
||||
const chainTail = this._claudeSessionChain[this._claudeSessionChain.length - 1];
|
||||
const candidates = [chainTail, options.resumeSessionId, this.id].filter((v): v is string => !!v);
|
||||
for (const candidate of candidates) {
|
||||
// A session Codeman DISCOVERED on the socket rather than created carries a
|
||||
// synthetic `restored-<fragment>` id, which fails claude's `uuid` token
|
||||
// pattern. The renderer would silently drop the resume flag and emit the
|
||||
// unpinned command, so say so here rather than letting the caller believe
|
||||
// the pane was pinned.
|
||||
if (resumeIdPattern?.type === 'token' && !matchesPattern(resumeIdPattern.pattern, candidate)) {
|
||||
console.log(`[Session] Not pinning resume id ${candidate} for relaunch: the CLI cannot accept that id shape`);
|
||||
continue;
|
||||
}
|
||||
if (!(await claudeTranscriptExists(candidate, configDir))) continue;
|
||||
options.resumeSessionId = candidate;
|
||||
return options;
|
||||
}
|
||||
// Nothing on disk to collide with, so the bare `--session-id <this.id>` the
|
||||
// unpinned options already carry is the correct command.
|
||||
return options;
|
||||
}
|
||||
|
||||
/** The session's Claude config dir when it has been relocated (#255), else undefined. */
|
||||
private _claudeConfigDir(): string | undefined {
|
||||
// Trimmed like `claudeProjectsDir()` trims the process-wide override: the
|
||||
// envOverrides schema validates keys only, and a whitespace-only value would
|
||||
// otherwise resolve to a relative path and read "no transcript" for everything.
|
||||
return this._envOverrides?.CLAUDE_CONFIG_DIR?.trim() || undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Force the custom-model selection's `launchModel` (pi/omp `custom/<id>`, grok's
|
||||
* `[model.<name>]` block name) onto the CLI's `model` launch param. Where that param
|
||||
@@ -2157,6 +2532,9 @@ export class Session extends EventEmitter {
|
||||
if (this.ptyProcess) {
|
||||
throw new Error('Session already has a running process');
|
||||
}
|
||||
if (this._closing) {
|
||||
throw new Error('Session is being closed');
|
||||
}
|
||||
|
||||
// Bounds the workspace-trust scan (see _maybeAcceptTrustDialog). Stamped here
|
||||
// rather than at PTY spawn so a slow mux attach still counts as startup.
|
||||
@@ -2265,7 +2643,7 @@ export class Session extends EventEmitter {
|
||||
// If mux wrapping is enabled, create or attach to a mux session
|
||||
if (this._useMux && this._mux) {
|
||||
try {
|
||||
const { isRestored } = await this._setupOrAttachMuxSession({
|
||||
const { isRestored, respawnedResumeId, respawnedDeadPane } = await this._setupOrAttachMuxSession({
|
||||
// Single source of truth shared with reattachRemote() (COD-108).
|
||||
respawnPaneOptions: this._buildRespawnPaneOptions(),
|
||||
createSessionOptions: {
|
||||
@@ -2273,6 +2651,7 @@ export class Session extends EventEmitter {
|
||||
workingDir: this.workingDir,
|
||||
mode: this.mode,
|
||||
name: this._name,
|
||||
cliName: this.cliPinnedName,
|
||||
niceConfig: this._niceConfig,
|
||||
model: this._model,
|
||||
claudeMode: this._claudeMode,
|
||||
@@ -2312,7 +2691,18 @@ export class Session extends EventEmitter {
|
||||
// persisted chain's tail is that conversation, reported first-hand by
|
||||
// the CLI's own hook, so it outranks every fallback here. A NEW pane has
|
||||
// an empty chain and falls through to the resume/alias fallbacks.
|
||||
restoredConversation = isRestored ? this._claudeSessionChain[this._claudeSessionChain.length - 1] : undefined;
|
||||
//
|
||||
// A dead-pane respawn is NOT that case for a CLI whose relaunch the resume
|
||||
// pin walk governs (`launch.chain === 'fallback'`): the CLI did stop, and
|
||||
// the walk may have passed over a chain tail with no transcript behind it,
|
||||
// so the conversation is whatever the respawn actually resumed. Undefined
|
||||
// there means the pane launched unpinned, which the fallbacks below name.
|
||||
const pinGovernsRespawn = respawnedDeadPane && getCli(this.mode)?.launch.chain === 'fallback';
|
||||
restoredConversation = pinGovernsRespawn
|
||||
? respawnedResumeId
|
||||
: isRestored
|
||||
? this._claudeSessionChain[this._claudeSessionChain.length - 1]
|
||||
: undefined;
|
||||
this._claudeSessionId =
|
||||
restoredConversation ||
|
||||
this._resumeSessionId ||
|
||||
@@ -2400,7 +2790,7 @@ export class Session extends EventEmitter {
|
||||
this._model,
|
||||
this._allowedTools,
|
||||
this._effort,
|
||||
this._name,
|
||||
this.cliPinnedName,
|
||||
getClaudeCliVersion()
|
||||
);
|
||||
this.ptyProcess = spawnPtyWithHelperRepair(() =>
|
||||
@@ -2413,6 +2803,7 @@ export class Session extends EventEmitter {
|
||||
env: { ...buildClaudeEnv(this.id), ...(this._envOverrides ?? {}) },
|
||||
})
|
||||
);
|
||||
this._notePtySpawnGeometry(120, 40);
|
||||
} catch (spawnErr) {
|
||||
console.error('[Session] Failed to spawn Claude PTY:', spawnErr);
|
||||
this._status = 'stopped';
|
||||
@@ -2705,10 +3096,82 @@ export class Session extends EventEmitter {
|
||||
if (now - this._lastPaneProbeAt < PANE_PROBE_MIN_INTERVAL_MS) return this._lastPaneProbeWorking;
|
||||
this._lastPaneProbeAt = now;
|
||||
const text = this._mux.capturePaneText?.(this._muxSession.muxName) ?? null;
|
||||
this._lastPaneProbeWorking = text === null ? null : this._workingLinePattern().test(text);
|
||||
// A turn that ended by handing off to workers the CLI waits for is work too: the
|
||||
// composer is up and the pane is quiet, but the next turn starts without the user.
|
||||
this._lastPaneProbeWorking =
|
||||
text === null ? null : this._workingLinePattern().test(text) || this._paneAwaitsWorkers(text);
|
||||
this._readWatching(text);
|
||||
return this._lastPaneProbeWorking;
|
||||
}
|
||||
|
||||
/**
|
||||
* Read the background-work chip off the same capture the working probe just took.
|
||||
*
|
||||
* The two questions are different. A turn that is running is work the user is waiting
|
||||
* for; a monitor, a backgrounded shell or a cloud session the agent started is work
|
||||
* the AGENT is waiting for, and it is the reason a pane can sit at its composer with
|
||||
* nothing to say and still not want anything from the user. `_confirmIdle` takes this
|
||||
* capture at exactly the moment the turn ends, which is the moment the answer starts
|
||||
* mattering.
|
||||
*
|
||||
* A capture that could not be read CLEARS the label rather than keeping the last one.
|
||||
* The two wrong answers are not symmetric: a stale label opens the next idle prompt
|
||||
* already acknowledged, so a failed capture would silence a real alert, while a dropped
|
||||
* label only costs a card and an alert that the next readable capture takes back.
|
||||
* Degrading toward the alert is the rule the whole signal is built on.
|
||||
*/
|
||||
private _readWatching(paneText: string | null): void {
|
||||
const pattern = this._watchingLinePattern();
|
||||
if (!pattern) return;
|
||||
// `null` is "the screen could not be read". That is no evidence either way, so the
|
||||
// label falls to null (and the change is announced below like any other).
|
||||
const label = paneText === null ? null : watchingLabel(paneText, pattern, this._watchingWindow);
|
||||
if (label === this._watching) return;
|
||||
this._watching = label;
|
||||
// ⚠️ This CHANGES while the session's status does not, so it needs an event of its
|
||||
// own. The label is usually set on the idle transition, which broadcasts anyway, but
|
||||
// it CLEARS when the work ends — and for a CLI whose background work ends without
|
||||
// taking a turn (measured on codex: a background terminal finishing repaints the row
|
||||
// away and nothing else happens) the session is idle before and after. Without this,
|
||||
// the server knew the badge was gone and every open page went on drawing it until
|
||||
// some unrelated event arrived.
|
||||
this.emit('watchingChanged');
|
||||
}
|
||||
|
||||
/**
|
||||
* The regex matching this CLI's background-work chip, or null for a CLI whose registry
|
||||
* entry declares none. Compiled once per session, like the working-line pattern, and
|
||||
* null rather than a fallback: no other CLI has been measured drawing such a chip, and
|
||||
* guessing one would badge sessions on the strength of an unread screen.
|
||||
*/
|
||||
private _watchingLinePattern(): RegExp | null {
|
||||
if (this._watchingLineRe === undefined) {
|
||||
// The pattern and the window it runs over are one decision, so they are resolved
|
||||
// together: how far up the screen a CLI's row can sit is as much a property of its
|
||||
// layout as the row itself. Claude writes on the last row and keeps the default,
|
||||
// Codex pins one above its composer and declares more.
|
||||
const detect = getCli(this.mode)?.capabilities.workDetect;
|
||||
this._watchingLineRe = detect?.watchingLine ? compileVersionRegex(detect.watchingLine) : null;
|
||||
this._watchingWindow = detect?.watchingLines ?? WATCHING_TAIL_LINES;
|
||||
}
|
||||
return this._watchingLineRe;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the newest turn on this screen ended waiting for workers the CLI started
|
||||
* (Claude's `✻ Waiting for 1 dynamic workflow to finish`). False for a CLI whose
|
||||
* registry entry declares no `awaitingLine`. See `isAwaitingWorkers()`.
|
||||
*/
|
||||
private _paneAwaitsWorkers(paneText: string): boolean {
|
||||
if (this._awaitingLineRe === undefined) {
|
||||
const src = getCli(this.mode)?.capabilities.workDetect?.awaitingLine;
|
||||
this._awaitingLineRe = src ? compileVersionRegex(src) : null;
|
||||
}
|
||||
if (!this._awaitingLineRe) return false;
|
||||
const glyph = getCli(this.mode)?.capabilities.workDetect?.promptGlyph ?? '❯';
|
||||
return isAwaitingWorkers(paneText, this._awaitingLineRe, glyph);
|
||||
}
|
||||
|
||||
/**
|
||||
* The regex matching this CLI's "a turn is running" status line.
|
||||
*
|
||||
@@ -2911,6 +3374,9 @@ export class Session extends EventEmitter {
|
||||
if (this.ptyProcess) {
|
||||
throw new Error('Session already has a running process');
|
||||
}
|
||||
if (this._closing) {
|
||||
throw new Error('Session is being closed');
|
||||
}
|
||||
|
||||
this._resetBuffers();
|
||||
|
||||
@@ -2981,6 +3447,7 @@ export class Session extends EventEmitter {
|
||||
env: buildShellEnv(this.id),
|
||||
})
|
||||
);
|
||||
this._notePtySpawnGeometry(120, 40);
|
||||
} catch (spawnErr) {
|
||||
console.error('[Session] Failed to spawn shell PTY:', spawnErr);
|
||||
this._status = 'stopped';
|
||||
@@ -3090,6 +3557,7 @@ export class Session extends EventEmitter {
|
||||
env: { ...buildClaudeEnv(this.id), ...(this._envOverrides ?? {}) },
|
||||
})
|
||||
);
|
||||
this._notePtySpawnGeometry(120, 40);
|
||||
} catch (spawnErr) {
|
||||
console.error('[Session] Failed to spawn Claude PTY for runPrompt:', spawnErr);
|
||||
this.emit(
|
||||
@@ -3190,6 +3658,9 @@ export class Session extends EventEmitter {
|
||||
}
|
||||
|
||||
private _clearAllTimers(): void {
|
||||
// Stop re-sending Enter for a prompt this session will never take now
|
||||
this._submitVerifier?.cancel();
|
||||
this._submitVerifier = null;
|
||||
// Clear the workspace-trust follow-up read
|
||||
if (this._trustDialogTimer) {
|
||||
clearTimeout(this._trustDialogTimer);
|
||||
@@ -3721,7 +4192,10 @@ export class Session extends EventEmitter {
|
||||
const submittedPrompt = this._trackSubmit(data, options);
|
||||
if (this._mux && this._muxSession) {
|
||||
const sent = await this._mux.sendInput(this.id, data);
|
||||
if (sent) this._emitSubmittedPrompt(submittedPrompt);
|
||||
if (sent) {
|
||||
this._emitSubmittedPrompt(submittedPrompt);
|
||||
this._verifySubmitted(data);
|
||||
}
|
||||
return sent;
|
||||
}
|
||||
// Fallback to PTY write
|
||||
@@ -3733,10 +4207,87 @@ export class Session extends EventEmitter {
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Arm the composer check for a prompt that went out some other way than
|
||||
* `writeViaMux`, e.g. cron's paste mode, which writes the body raw and its Enter
|
||||
* separately. `text` is what the composer line starts with while the prompt is still
|
||||
* unsent; the check re-presses Enter only while that holds.
|
||||
*/
|
||||
verifySubmitted(text: string): void {
|
||||
this._verifySubmitted(`${text}\r`);
|
||||
}
|
||||
|
||||
/**
|
||||
* Arm the composer check for a write that carried Enter (session-submit-verifier.ts):
|
||||
* Claude Code 2.1.277+ ignores Enter for the first 30-50 s after the composer paints,
|
||||
* so the pair `sendInput` just sent can leave the text stranded. Only a mux session
|
||||
* can read its pane, only text can be stranded, and the glyph is the CLI's own.
|
||||
*/
|
||||
private _verifySubmitted(data: string): void {
|
||||
if (!data.includes('\r') || !this._mux?.capturePaneText || !this._muxSession) return;
|
||||
const text = data.replace(/[\r\n]/g, '').trimEnd();
|
||||
if (!text) return;
|
||||
this._submitVerifier ??= new SubmitVerifier({
|
||||
capture: () =>
|
||||
this._isStopped || !this._mux || !this._muxSession
|
||||
? null
|
||||
: this._mux.capturePaneText?.(this._muxSession.muxName),
|
||||
sendEnter: () => this._mux?.sendInput(this.id, '\r'),
|
||||
// ⚠ NO fallback glyph here, unlike the screen-reading probe elsewhere in this file.
|
||||
// Only claude and codex declare a promptGlyph; the other eight modes would fall back
|
||||
// to claude's `❯`, which is ALSO starship's default shell prompt (and pure's, and
|
||||
// spaceship's, and p10k lean's). On a shell session the line `❯ npm run build` sits
|
||||
// on screen for as long as the command runs, promptStillInComposer() reads that as
|
||||
// "still unsubmitted", and the verifier presses Enter into the running program's
|
||||
// stdin on its 2s..60s schedule. Mostly a stray blank line; not harmless against a
|
||||
// y/N prompt, `read -p`, an installer or a pager, where it takes the default.
|
||||
// promptStillInComposer() returns undefined for an empty glyph, so this makes the
|
||||
// verifier inert for every CLI that does not declare one, which is what the Claude
|
||||
// Code 2.1.277 defect it exists for actually calls for.
|
||||
glyph: () => getCli(this.mode)?.capabilities.workDetect?.promptGlyph ?? '',
|
||||
log: (m) => console.log(`[Session ${this.id.slice(0, 8)}] ${m}`),
|
||||
});
|
||||
this._submitVerifier.arm(text);
|
||||
}
|
||||
|
||||
/** Current PTY dimensions — used to skip no-op resizes that trigger Ink redraws */
|
||||
private _ptyCols = 120;
|
||||
private _ptyRows = 40;
|
||||
|
||||
/**
|
||||
* Record the geometry a PTY was just spawned at. A reattached pane keeps the
|
||||
* tmux window's size, not the constructor's 120x40, and without this
|
||||
* `ptyGeometry` reported the old numbers for a live pane and the dedupe in
|
||||
* `resize()` skipped a real resize that happened to match them.
|
||||
*/
|
||||
private _notePtySpawnGeometry(cols: number, rows: number): void {
|
||||
this._ptyCols = cols;
|
||||
this._ptyRows = rows;
|
||||
}
|
||||
|
||||
/**
|
||||
* The geometry the CLI is actually drawing for, or null when nothing is
|
||||
* drawing.
|
||||
*
|
||||
* Exposed because `resize()` can decline a request outright (arbitration
|
||||
* below) and the asking client has no other way to find out: a browser
|
||||
* terminal that keeps a shape the PTY refused renders garbled output rather
|
||||
* than wrong-sized output, because Claude Code's repaints are computed from
|
||||
* the width it was told (issue #464). Both transports report this back.
|
||||
*
|
||||
* ⚠️ NULL WITHOUT A PANE, never the field values. The fields are seeded at
|
||||
* spawn (`_notePtySpawnGeometry`) and moved by `resize()`, but a session with
|
||||
* a dead pane (or one created through the API and never started) still
|
||||
* holds the constructor defaults of 120x40, or the size of a pane that is
|
||||
* gone. Reporting those made a client adopt a size no process had ever
|
||||
* been told, and on anything narrower than 120 columns it claimed another
|
||||
* device owned the pane when none existed. `reconcilePtyGeometry` treats a
|
||||
* report with no finite numbers as no evidence, which is the truth here.
|
||||
*/
|
||||
get ptyGeometry(): { cols: number; rows: number } | null {
|
||||
return this.ptyProcess ? { cols: this._ptyCols, rows: this._ptyRows } : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Live WebSocket connections that have announced a desktop viewport for this
|
||||
* session. While at least one is registered, small-viewport (mobile/tablet)
|
||||
@@ -3822,6 +4373,10 @@ export class Session extends EventEmitter {
|
||||
}
|
||||
if (isSmallViewport && this._desktopSizeClaims.size > 0) {
|
||||
if (Date.now() - this._lastDesktopActivityAt < Session.DESKTOP_CLAIM_IDLE_MS) {
|
||||
// Declined. The caller is told nothing here on purpose — the decision
|
||||
// belongs to the session, not the socket — but the caller MUST report
|
||||
// `ptyGeometry` back afterwards so the asking client can adopt
|
||||
// the shape it did not get. Both transports do; see issue #464.
|
||||
return;
|
||||
}
|
||||
this._mobileSizeOverride = true;
|
||||
@@ -3921,6 +4476,9 @@ export class Session extends EventEmitter {
|
||||
async stop(killMux: boolean = true): Promise<void> {
|
||||
// Set stopped flag first to prevent new timers from being created
|
||||
this._isStopped = true;
|
||||
// A pane that is gone is watching nothing. Nothing probes a stopped session, so
|
||||
// without this the last chip it drew would ride along on its row forever.
|
||||
this._watching = null;
|
||||
|
||||
this._clearAllTimers();
|
||||
|
||||
|
||||
+404
-19
@@ -58,6 +58,7 @@ import {
|
||||
type SessionRemote,
|
||||
type SessionDocker,
|
||||
type DockerCommandMode,
|
||||
type PaneExit,
|
||||
} from './types.js';
|
||||
import { getCli } from './config/cli-registry/registry.js';
|
||||
import { missingCliMessage, resolveCliBinDir } from './utils/cli-resolver.js';
|
||||
@@ -152,6 +153,16 @@ const GRACEFUL_SHUTDOWN_WAIT_MS = 100;
|
||||
/** Default stats collection interval (2 seconds) */
|
||||
const DEFAULT_STATS_INTERVAL_MS = 2000;
|
||||
|
||||
/**
|
||||
* How often the pane-exit watcher re-reads every pane on the socket. The
|
||||
* watcher owns this cadence: it does NOT ride `startStatsCollection()`, whose
|
||||
* lifetime a browser panel controls (see {@link TmuxManager.startPaneExitWatcher}).
|
||||
* Matched to the stats cadence above because both cost one batched tmux read.
|
||||
* ⚠ It does NOT bound a read: EXEC_TIMEOUT_MS is 5000 ms, so a slow read can
|
||||
* outlive two ticks, which is exactly why `paneExitReadInFlight` exists.
|
||||
*/
|
||||
const DEFAULT_PANE_EXIT_INTERVAL_MS = 2000;
|
||||
|
||||
/** Default remote-reconnect watcher poll interval (5 seconds) — COD-108 */
|
||||
const DEFAULT_REMOTE_RECONNECT_INTERVAL_MS = 5000;
|
||||
|
||||
@@ -219,8 +230,18 @@ const DEFAULT_CODEMAN_TMUX_SOCKET = DEFAULT_TMUX_SOCKET;
|
||||
*/
|
||||
const PANE_LIST_SEP = '|';
|
||||
|
||||
/** Format string for `tmux list-panes -F`. Keep in sync with {@link parsePaneList}. */
|
||||
const PANE_LIST_FORMAT = `#{session_name}${PANE_LIST_SEP}#{pane_pid}`;
|
||||
/**
|
||||
* Format string for `tmux list-panes -F`. Keep in sync with {@link parsePaneRows}.
|
||||
*
|
||||
* The three `pane_dead*` fields carry the agent-exit signal of Ark0N/Codeman#446.
|
||||
* Appending them is backward compatible in both directions. A tmux that does not
|
||||
* know a variable substitutes the empty string rather than failing, which is how
|
||||
* tmux 3.2a answers `#{pane_dead_signal}` (added in 3.4), and the parser reads a
|
||||
* short row as "pid known, deadness unknown" rather than discarding it.
|
||||
*/
|
||||
const PANE_LIST_FORMAT =
|
||||
`#{session_name}${PANE_LIST_SEP}#{pane_pid}` +
|
||||
`${PANE_LIST_SEP}#{pane_dead}${PANE_LIST_SEP}#{pane_dead_status}${PANE_LIST_SEP}#{pane_dead_signal}`;
|
||||
|
||||
/**
|
||||
* 构建 pane 启动前的 nofile 修复命令。
|
||||
@@ -235,26 +256,154 @@ export function buildNofileLimitCommand(targetLimit = CLAUDE_CODE_NOFILE_LIMIT):
|
||||
return `ulimit -Sn ${safeLimit} 2>/dev/null || ulimit -n ${safeLimit} 2>/dev/null || true`;
|
||||
}
|
||||
|
||||
/** One pane of one tmux session, as {@link parsePaneRows} reads it off the wire. */
|
||||
export interface PaneRow {
|
||||
/** tmux session this pane belongs to. Repeats once per pane of a split session. */
|
||||
sessionName: string;
|
||||
/** `#{pane_pid}` — the process tmux started in the pane. */
|
||||
pid: number;
|
||||
/** `#{pane_dead}` — true for 1, false for 0, undefined when tmux said nothing. */
|
||||
dead?: boolean;
|
||||
/** `#{pane_dead_status}` — the exit code, absent when tmux reported none. */
|
||||
exitStatus?: number;
|
||||
/** `#{pane_dead_signal}` — the killing signal, absent before tmux 3.4 and when unsignalled. */
|
||||
exitSignal?: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse the output of `tmux list-panes -a -F '#{session_name}|#{pane_pid}'`
|
||||
* into a Map of session-name → pane pid. Exported for unit testing.
|
||||
* One pane-exit reading, with the pane pid that produced it.
|
||||
*
|
||||
* The pid never leaves this module. It is what distinguishes "the same dead
|
||||
* pane, seen again" from "a second command in the same pane that also exited
|
||||
* with the same status", so the `at` stamp can hold across the first and must
|
||||
* not across the second. {@link PaneExit} itself stays free of it: the pid on
|
||||
* the session record is the attach client's, and a second pid there would
|
||||
* invite exactly the confusion Ark0N/Codeman#446 is about.
|
||||
*/
|
||||
export interface PaneExitObservation {
|
||||
/** `#{pane_pid}` of the pane this reading came from. */
|
||||
panePid: number;
|
||||
/** What to publish on the session record. */
|
||||
exit: PaneExit;
|
||||
}
|
||||
|
||||
/**
|
||||
* A {@link PaneExitObservation} as the manager stores it, with a count of the
|
||||
* authoritative reads that have seen this same exit. The count is what lets
|
||||
* the exited-agent sweep act only on a death that more than one read agreed on
|
||||
* (`CLEAN_EXIT_CONFIRMING_READS` in `pane-exit-sweep.ts`). A failed or skipped
|
||||
* read never reaches {@link TmuxManager.applyPaneExits}, so it neither raises
|
||||
* the count nor resets it.
|
||||
*/
|
||||
interface TrackedPaneExit extends PaneExitObservation {
|
||||
/** Authoritative reads that saw this exit, counting the first. */
|
||||
reads: number;
|
||||
}
|
||||
|
||||
/** Read one optional numeric field; a blank or non-numeric value is "not reported". */
|
||||
function paneField(fields: string[], index: number): number | undefined {
|
||||
const raw = fields[index];
|
||||
if (raw === undefined || raw === '') return undefined;
|
||||
const value = parseInt(raw, 10);
|
||||
return Number.isNaN(value) ? undefined : value;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse the output of `tmux list-panes -a -F` under {@link PANE_LIST_FORMAT}
|
||||
* into one row per pane, in tmux's own order. Exported for unit testing.
|
||||
*
|
||||
* - Skips empty lines and lines without the separator.
|
||||
* - Skips entries with a non-numeric pid or empty name.
|
||||
* - Leaves every field after the pid undefined when it is blank or absent, so a
|
||||
* row from an older tmux still yields its pid.
|
||||
*/
|
||||
export function parsePaneList(output: string): Map<string, number> {
|
||||
const result = new Map<string, number>();
|
||||
export function parsePaneRows(output: string): PaneRow[] {
|
||||
const rows: PaneRow[] = [];
|
||||
for (const line of output.split('\n')) {
|
||||
if (!line) continue;
|
||||
const sep = line.indexOf(PANE_LIST_SEP);
|
||||
if (sep === -1) continue;
|
||||
const name = line.slice(0, sep);
|
||||
const pid = parseInt(line.slice(sep + 1), 10);
|
||||
if (name && !Number.isNaN(pid)) {
|
||||
result.set(name, pid);
|
||||
}
|
||||
if (!line.includes(PANE_LIST_SEP)) continue;
|
||||
const fields = line.split(PANE_LIST_SEP);
|
||||
const sessionName = fields[0];
|
||||
const pid = parseInt(fields[1] ?? '', 10);
|
||||
if (!sessionName || Number.isNaN(pid)) continue;
|
||||
const deadFlag = fields[2];
|
||||
rows.push({
|
||||
sessionName,
|
||||
pid,
|
||||
dead: deadFlag === '1' ? true : deadFlag === '0' ? false : undefined,
|
||||
exitStatus: paneField(fields, 3),
|
||||
exitSignal: paneField(fields, 4),
|
||||
});
|
||||
}
|
||||
return result;
|
||||
return rows;
|
||||
}
|
||||
|
||||
/**
|
||||
* Decide, from every pane tmux listed, which tmux sessions have an exited agent.
|
||||
* Exported for unit testing. Returns one entry per session with a known answer;
|
||||
* a session absent from the map is UNKNOWN, which must never render as alive.
|
||||
*
|
||||
* Two rules make a positive answer trustworthy:
|
||||
*
|
||||
* A session answers only when tmux listed EXACTLY ONE pane for it. Codeman
|
||||
* creates one pane per session and `isPaneDead()` reads one pane, so a session
|
||||
* the user has split by hand has no single "the agent" to report on, and
|
||||
* guessing which of its panes speaks for the session could report a live
|
||||
* session as exited.
|
||||
*
|
||||
* A pane answers only when `#{pane_dead}` said 1 or 0. An empty field is a tmux
|
||||
* that did not answer, not a live pane.
|
||||
*
|
||||
* `status` and `signal` stay absent when tmux did not report them. Measured on
|
||||
* tmux 3.2a, a SIGKILLed pane reports neither, so folding an absent status into
|
||||
* 0 would turn an unexplained death into a clean exit.
|
||||
*/
|
||||
export function derivePaneExits(rows: PaneRow[], now: number): Map<string, PaneExitObservation> {
|
||||
const panesPerSession = new Map<string, number>();
|
||||
for (const row of rows) {
|
||||
panesPerSession.set(row.sessionName, (panesPerSession.get(row.sessionName) ?? 0) + 1);
|
||||
}
|
||||
const exits = new Map<string, PaneExitObservation>();
|
||||
for (const row of rows) {
|
||||
if (panesPerSession.get(row.sessionName) !== 1) continue;
|
||||
if (row.dead !== true) continue;
|
||||
exits.set(row.sessionName, {
|
||||
panePid: row.pid,
|
||||
exit: {
|
||||
...(row.exitStatus !== undefined ? { status: row.exitStatus } : {}),
|
||||
...(row.exitSignal !== undefined ? { signal: row.exitSignal } : {}),
|
||||
at: now,
|
||||
},
|
||||
});
|
||||
}
|
||||
return exits;
|
||||
}
|
||||
|
||||
/**
|
||||
* Could any of these tmux sessions ever produce a pane-exit answer? Exported
|
||||
* for unit testing.
|
||||
*
|
||||
* Mirrors `Session.paneExitApplies`, which is where the rule is enforced. A
|
||||
* remote session's local pane holds the ssh client, a docker case's holds a
|
||||
* `docker exec` into the container's own tmux, and a record rebuilt from the
|
||||
* socket carries no provenance at all, so the session end forces all three to
|
||||
* UNKNOWN whatever tmux reports. A tick that sees only those has nothing to
|
||||
* learn, and `refreshPaneExits()` skips its tmux read rather than paying for
|
||||
* the answer.
|
||||
*
|
||||
* ⚠ This gates the READ, never the watcher. The watcher is always-on by
|
||||
* design (see {@link TmuxManager.startPaneExitWatcher}), so it keeps ticking
|
||||
* with nothing to observe and picks the read straight back up as soon as one
|
||||
* local session exists.
|
||||
*/
|
||||
export function hasObservablePaneSession(sessions: Iterable<MuxSession>): boolean {
|
||||
for (const session of sessions) {
|
||||
if (session.remote) continue;
|
||||
if (session.docker) continue;
|
||||
if (session.discovered === true) continue;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -735,7 +884,7 @@ export function buildSpawnCommand(options: {
|
||||
effort?: EffortLevel;
|
||||
/** Resolved by resolveStatusLineCliCommand (hooks-config.ts) — undefined skips the exporter. Claude only. */
|
||||
statusLineCommand?: string;
|
||||
/** Codeman session name, passed to claude as `--name` (version-gated, sanitized; local spawns only). */
|
||||
/** Name pinned on claude as `--name` (version-gated, sanitized; local spawns only). Only a user-chosen name: see `Session.cliPinnedName`. */
|
||||
sessionName?: string;
|
||||
/**
|
||||
* Claude CLI version for the `--name` gate. Omitted = probe the local CLI
|
||||
@@ -1527,6 +1676,30 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
private mouseSyncInterval: NodeJS.Timeout | null = null;
|
||||
/** Track last-known pane count per session to avoid unnecessary tmux set-option calls */
|
||||
private lastPaneCount: Map<string, number> = new Map();
|
||||
/**
|
||||
* muxName → the exited agent the pane-exit watcher last observed
|
||||
* (Ark0N/Codeman#446). Absence is the UNKNOWN arm of the tri-state, so an
|
||||
* entry goes the moment tmux stops reporting the pane dead, and the map is
|
||||
* empty until the first read runs. The manager reports what tmux says and
|
||||
* nothing more: the scoping that hides this for remote and docker sessions
|
||||
* lives on `Session`, because the remote-reconnect watcher above needs the
|
||||
* raw pane reading.
|
||||
*/
|
||||
private paneExits: Map<string, TrackedPaneExit> = new Map();
|
||||
/** The pane-exit watcher's own interval. Runs whether or not stats are on. */
|
||||
private paneExitInterval: NodeJS.Timeout | null = null;
|
||||
/**
|
||||
* True while a pane read is in flight. `EXEC_TIMEOUT_MS` is 5000 ms against a
|
||||
* poll interval of 2000 ms, so without this a slow read overlaps the next two
|
||||
* and the older one can resolve last and win.
|
||||
*/
|
||||
private paneExitReadInFlight = false;
|
||||
/**
|
||||
* Bumped by every deliberate {@link clearPaneExit}. A read that started before
|
||||
* a clear carries the older generation and is discarded rather than writing
|
||||
* the death back over the pane that has just replaced it.
|
||||
*/
|
||||
private paneExitGeneration = 0;
|
||||
|
||||
// ── COD-108 remote-reconnect watcher state ────────────────────────────────
|
||||
/** Periodic watcher that re-establishes dropped remote sessions. */
|
||||
@@ -1894,6 +2067,7 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
workingDir,
|
||||
mode,
|
||||
name,
|
||||
cliName,
|
||||
niceConfig,
|
||||
model,
|
||||
claudeMode,
|
||||
@@ -1997,7 +2171,7 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
resumeSessionId,
|
||||
effort,
|
||||
statusLineCommand,
|
||||
sessionName: name,
|
||||
sessionName: cliName,
|
||||
});
|
||||
|
||||
const config = niceConfig || DEFAULT_NICE_CONFIG;
|
||||
@@ -2225,7 +2399,7 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
effort,
|
||||
remote,
|
||||
docker,
|
||||
name,
|
||||
cliName,
|
||||
} = options;
|
||||
const session = this.sessions.get(sessionId);
|
||||
if (!session) return null;
|
||||
@@ -2262,7 +2436,7 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
resumeSessionId,
|
||||
effort,
|
||||
statusLineCommand,
|
||||
sessionName: name,
|
||||
sessionName: cliName,
|
||||
});
|
||||
const config = niceConfig || DEFAULT_NICE_CONFIG;
|
||||
const cmd = wrapWithNice(baseCmd, config);
|
||||
@@ -2297,6 +2471,11 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
);
|
||||
// Wait for the respawned process to start
|
||||
await new Promise((resolve) => setTimeout(resolve, TMUX_CREATION_WAIT_MS));
|
||||
// The pane now runs a fresh command, so whatever the last read observed of
|
||||
// the old one is history. Clearing it here rather than waiting for the next
|
||||
// poll also invalidates any read already in flight, which would otherwise
|
||||
// write the old death back over the pane that just replaced it.
|
||||
this.clearPaneExit(muxName);
|
||||
const pid = this.getPanePid(muxName);
|
||||
if (pid) session.pid = pid;
|
||||
return pid;
|
||||
@@ -2508,6 +2687,7 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
}
|
||||
}
|
||||
this.lastPaneCount.delete(session.muxName);
|
||||
this.clearPaneExit(session.muxName);
|
||||
this.sessions.delete(sessionId);
|
||||
this.clearRemoteReconnectState(sessionId);
|
||||
this.saveSessions();
|
||||
@@ -2618,6 +2798,7 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
}
|
||||
|
||||
this.lastPaneCount.delete(session.muxName);
|
||||
this.clearPaneExit(session.muxName);
|
||||
this.sessions.delete(sessionId);
|
||||
this.clearRemoteReconnectState(sessionId);
|
||||
this.saveSessions();
|
||||
@@ -2671,7 +2852,14 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
encoding: 'utf-8',
|
||||
timeout: EXEC_TIMEOUT_MS,
|
||||
}).trim();
|
||||
active = parsePaneList(output);
|
||||
const rows = parsePaneRows(output);
|
||||
active = new Map(rows.map((row) => [row.sessionName, row.pid]));
|
||||
// The same read answers both questions, so recovery starts with a pane-exit
|
||||
// reading rather than waiting for the first stats tick — which may never
|
||||
// come, since the collector only starts when boot found a live session.
|
||||
if (rows.length > 0) {
|
||||
this.applyPaneExits(derivePaneExits(rows, Date.now()));
|
||||
}
|
||||
} catch (err) {
|
||||
console.error('[TmuxManager] Failed to list tmux panes:', err);
|
||||
active = new Map();
|
||||
@@ -2686,6 +2874,7 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
} else {
|
||||
dead.push(sessionId);
|
||||
this.sessions.delete(sessionId);
|
||||
this.clearPaneExit(session.muxName);
|
||||
this.clearRemoteReconnectState(sessionId);
|
||||
this.emit('sessionDied', { sessionId });
|
||||
}
|
||||
@@ -2722,6 +2911,13 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
mode: 'claude',
|
||||
attached: false,
|
||||
name: `Restored: ${sessionName}`,
|
||||
// Every field above except the name and the pid is a guess: this record
|
||||
// was rebuilt from the socket because Codeman's own bookkeeping did not
|
||||
// have it. The synthetic id also cannot find the session's state.json
|
||||
// entry, so a remote or docker session rediscovered this way arrives
|
||||
// looking local. Consumers that would be wrong about such a session
|
||||
// read this flag and fail closed — see `Session.paneExitApplies`.
|
||||
discovered: true,
|
||||
};
|
||||
this.sessions.set(sessionId, session);
|
||||
knownMuxNames.add(sessionName);
|
||||
@@ -2882,6 +3078,185 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
}));
|
||||
}
|
||||
|
||||
/**
|
||||
* What the last pane read saw of this tmux session's agent. `undefined` is
|
||||
* the UNKNOWN answer and must never be rendered as "alive": it covers a pane
|
||||
* that is running, a tmux session that no longer exists, a probe that failed,
|
||||
* and every poll that has not run yet. See {@link PaneExit}.
|
||||
*/
|
||||
getPaneExit(muxName: string): PaneExit | undefined {
|
||||
return this.paneExits.get(muxName)?.exit;
|
||||
}
|
||||
|
||||
/**
|
||||
* How many authoritative pane reads have agreed on the exit that
|
||||
* {@link getPaneExit} reports, or 0 when it reports none. A new observation
|
||||
* starts at 1, and every later read that sees the same pane with the same
|
||||
* status and signal adds one.
|
||||
*/
|
||||
getPaneExitReadCount(muxName: string): number {
|
||||
return this.paneExits.get(muxName)?.reads ?? 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-read every pane on the socket and refresh {@link paneExits}. ONE batched
|
||||
* `tmux list-panes -a` answers for every session at once, which is why this
|
||||
* polls rather than probing per session.
|
||||
*
|
||||
* A failed or empty probe leaves the previous answers ALONE rather than
|
||||
* clearing them, because the two cannot be told apart: the command ends in
|
||||
* `|| true`, so a tmux that errored and a socket with genuinely no panes both
|
||||
* arrive as empty output. Treating that as "tmux did not answer" is the
|
||||
* conservative reading — clearing on it would turn a transient failure into a
|
||||
* silent retraction of a death Codeman had already observed, and the cost of
|
||||
* being wrong the other way is one stale entry for a socket that no longer
|
||||
* has the pane. A NON-empty read is different: `list-panes -a` lists
|
||||
* every pane on the socket, so it is authoritative and {@link applyPaneExits}
|
||||
* prunes against it.
|
||||
*
|
||||
* Two guards keep a slow read from undoing a fast one. A read already in
|
||||
* flight suppresses the next poll, and a read that started before a
|
||||
* {@link clearPaneExit} is discarded when it lands.
|
||||
*
|
||||
* A third guard skips the read entirely while no session on this manager
|
||||
* could produce an answer ({@link hasObservablePaneSession}). Skipping
|
||||
* retracts nothing, for the same reason a failed read does not: the map
|
||||
* still holds what the last real read saw, and every path that puts a new
|
||||
* command in a pane calls {@link clearPaneExit} itself.
|
||||
*/
|
||||
async refreshPaneExits(now: number = Date.now()): Promise<void> {
|
||||
// Nothing on this socket could answer, so do not read tmux to find that
|
||||
// out. See `hasObservablePaneSession`: the watcher above still ticks.
|
||||
if (!hasObservablePaneSession(this.sessions.values())) return;
|
||||
if (this.paneExitReadInFlight) return;
|
||||
|
||||
const generation = this.paneExitGeneration;
|
||||
this.paneExitReadInFlight = true;
|
||||
let rows: PaneRow[];
|
||||
try {
|
||||
rows = await this.readPaneRows();
|
||||
} finally {
|
||||
this.paneExitReadInFlight = false;
|
||||
}
|
||||
if (rows.length === 0) return;
|
||||
// A pane was respawned or killed while this read was out, so what it saw is
|
||||
// already history. Dropping it is what stops a freshly respawned pane from
|
||||
// being republished as exited.
|
||||
if (generation !== this.paneExitGeneration) return;
|
||||
|
||||
this.applyPaneExits(derivePaneExits(rows, now));
|
||||
}
|
||||
|
||||
/**
|
||||
* Read every pane on the socket. The ONLY part of the pane-exit watcher that
|
||||
* touches tmux, which is what lets a test subclass drive the guards in
|
||||
* {@link refreshPaneExits} — the in-flight suppression, the generation
|
||||
* check, the empty-read retraction rule and the read gate — against rows it
|
||||
* chooses. Split out for the reason `runRemoteReconnectTick` is: a guard no
|
||||
* test can reach is a guard that can be deleted without anything failing.
|
||||
*
|
||||
* A failed read answers with NO rows, which the caller treats as "tmux did
|
||||
* not answer" and which therefore retracts nothing.
|
||||
*/
|
||||
protected async readPaneRows(): Promise<PaneRow[]> {
|
||||
// The test-mode gate lives HERE rather than at the top of the tick, so that
|
||||
// what tests cannot do is spawn a process, not exercise the bookkeeping.
|
||||
if (IS_TEST_MODE) return [];
|
||||
try {
|
||||
// execAsync, not execSync: this runs on a 2000 ms timer, and a synchronous
|
||||
// exec freezes the port while the process stays alive (see the
|
||||
// event-loop-monitor note in CLAUDE.md). The three `isPaneDead()` callers
|
||||
// stay synchronous because each is answering one request right then.
|
||||
const { stdout } = await execAsync(`${this.tmux()} list-panes -a -F '${PANE_LIST_FORMAT}' 2>/dev/null || true`, {
|
||||
encoding: 'utf-8',
|
||||
timeout: EXEC_TIMEOUT_MS,
|
||||
});
|
||||
return parsePaneRows(stdout.trim());
|
||||
} catch (err) {
|
||||
console.error('[TmuxManager] Failed to read pane exit state:', err);
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fold one authoritative observation into {@link paneExits}. Split out from
|
||||
* the tmux call so the merge rules are unit-testable.
|
||||
*
|
||||
* `observed` comes from a read of EVERY pane on the socket, so a session
|
||||
* missing from it has no exit to report and its entry goes. That is what
|
||||
* keeps the map from growing without bound as tmux sessions come and go
|
||||
* outside `killSession()`. Only the caller may decide a read is authoritative:
|
||||
* a failed or empty one never reaches here.
|
||||
*
|
||||
* An entry keeps the `at` of the FIRST read that saw that exit, so the stamp
|
||||
* says when the agent was found gone rather than when the last poll ran. A
|
||||
* changed status, a changed signal, or a different pane pid all start a new
|
||||
* observation — the pid is what catches a second command in the same pane
|
||||
* that happened to exit the same way.
|
||||
*
|
||||
* The same rule decides the read count: a repeat of the stored exit adds one,
|
||||
* and anything that starts a new observation starts the count again at 1.
|
||||
*/
|
||||
applyPaneExits(observed: Map<string, PaneExitObservation>): void {
|
||||
for (const muxName of [...this.paneExits.keys()]) {
|
||||
if (!observed.has(muxName)) this.paneExits.delete(muxName);
|
||||
}
|
||||
for (const [muxName, next] of observed) {
|
||||
const prev = this.paneExits.get(muxName);
|
||||
const sameExit =
|
||||
prev !== undefined &&
|
||||
prev.panePid === next.panePid &&
|
||||
prev.exit.status === next.exit.status &&
|
||||
prev.exit.signal === next.exit.signal;
|
||||
this.paneExits.set(muxName, sameExit ? { ...prev, reads: prev.reads + 1 } : { ...next, reads: 1 });
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Forget a session's exit observation, e.g. once its pane has been respawned.
|
||||
* Also invalidates any read already in flight, so the answer this retracts
|
||||
* cannot be written back a moment later.
|
||||
*/
|
||||
clearPaneExit(muxName: string): void {
|
||||
this.paneExits.delete(muxName);
|
||||
this.paneExitGeneration++;
|
||||
}
|
||||
|
||||
/**
|
||||
* Poll for exited agents, on the manager's own interval.
|
||||
*
|
||||
* Deliberately NOT part of `startStatsCollection()`. That collector is armed
|
||||
* when the browser opens the Monitor panel and DISARMED when it closes it
|
||||
* (`panels-ui.js`), and it is skipped at boot entirely when no session was
|
||||
* recovered — so riding it would leave a session created on a freshly booted
|
||||
* server reporting nothing at all, and would let one browser turn exit
|
||||
* detection off for every other. Started unconditionally, like the mouse-mode
|
||||
* sync and the remote-reconnect watcher below.
|
||||
*
|
||||
* The `paneExitsUpdated` event is internal to the server; nothing here adds an
|
||||
* SSE event, and the field reaches the browser on `session:updated`.
|
||||
*/
|
||||
startPaneExitWatcher(intervalMs: number = DEFAULT_PANE_EXIT_INTERVAL_MS): void {
|
||||
if (this.paneExitInterval) {
|
||||
clearInterval(this.paneExitInterval);
|
||||
}
|
||||
this.paneExitInterval = setInterval(() => {
|
||||
// No IS_TEST_MODE guard: `readPaneRows()` is the only thing that would
|
||||
// spawn a process and it refuses under test, so a test can drive this
|
||||
// whole loop with fake timers instead of being locked out of it.
|
||||
void this.refreshPaneExits()
|
||||
.then(() => this.emit('paneExitsUpdated'))
|
||||
.catch((err) => console.error('[TmuxManager] Pane exit watcher error:', err));
|
||||
}, intervalMs);
|
||||
}
|
||||
|
||||
stopPaneExitWatcher(): void {
|
||||
if (this.paneExitInterval) {
|
||||
clearInterval(this.paneExitInterval);
|
||||
this.paneExitInterval = null;
|
||||
}
|
||||
}
|
||||
|
||||
startStatsCollection(intervalMs: number = DEFAULT_STATS_INTERVAL_MS): void {
|
||||
if (this.statsInterval) {
|
||||
clearInterval(this.statsInterval);
|
||||
@@ -3101,6 +3476,8 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
|
||||
destroy(): void {
|
||||
this.stopStatsCollection();
|
||||
this.stopPaneExitWatcher();
|
||||
this.paneExits.clear();
|
||||
this.stopMouseModeSync();
|
||||
this.stopRemoteReconnectWatcher();
|
||||
this.reconnectState.clear();
|
||||
@@ -3482,6 +3859,14 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
{ encoding: 'utf-8', timeout: EXEC_TIMEOUT_MS }
|
||||
)
|
||||
);
|
||||
// Report the size the pane was really drawing at. The visible-frame path
|
||||
// below addresses every row absolutely, so a consumer whose terminal is
|
||||
// shorter than this piles the overflow rows onto its last line and loses
|
||||
// the rows it overwrote. The full-history path instead ends in a RELATIVE
|
||||
// cursor move, which costs it nothing when the two sizes disagree, so the
|
||||
// geometry is reported there for diagnosis rather than for repair. Only
|
||||
// the caller can see both sizes, so hand it this one.
|
||||
if (opts && geometry) opts.capturedGeometry = { cols: geometry.cols, rows: geometry.rows };
|
||||
|
||||
if (fullHistory) {
|
||||
// Without geometry there is no cursor move, so fall back to the old trim.
|
||||
|
||||
@@ -28,8 +28,11 @@
|
||||
import type { ApprovalItem, ApprovalOption } from '../web/approval-inbox.js';
|
||||
import type { TuiApprovalAnswer } from './tui-client.js';
|
||||
|
||||
/** Card severity, in the same red/yellow vocabulary the web inbox uses. */
|
||||
export type TuiApprovalTone = 'err' | 'warn';
|
||||
/**
|
||||
* Card severity, in the same red/yellow vocabulary the web inbox uses, plus the quiet
|
||||
* third case: an item that opened acknowledged asks for nothing and reads grey.
|
||||
*/
|
||||
export type TuiApprovalTone = 'err' | 'warn' | 'info';
|
||||
|
||||
export interface TuiApprovalCard {
|
||||
tone: TuiApprovalTone;
|
||||
@@ -50,8 +53,14 @@ function clean(text: string | undefined): string {
|
||||
return (text ?? '').replace(/\s+/g, ' ').trim().slice(0, MAX_CARD_TEXT);
|
||||
}
|
||||
|
||||
/**
|
||||
* How loud the card is. An idle prompt the inbox opened ALREADY acknowledged is not
|
||||
* asking for anything — the session is watching work it started itself — so it drops to
|
||||
* `info` and out of the warning vocabulary the other two share with the web inbox.
|
||||
*/
|
||||
export function approvalTone(item: ApprovalItem): TuiApprovalTone {
|
||||
return item.kind === 'idle' ? 'warn' : 'err';
|
||||
if (item.kind !== 'idle') return 'err';
|
||||
return item.acknowledgedReason ? 'info' : 'warn';
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -65,9 +74,13 @@ export function approvalCard(item: ApprovalItem): TuiApprovalCard {
|
||||
const summary = clean(item.toolSummary) || clean(item.toolName);
|
||||
|
||||
if (item.kind === 'idle') {
|
||||
// Say what it is waiting for rather than asking for a reply, in the same words the
|
||||
// web drawer uses for the same item. The prompt is still answerable, so the hint
|
||||
// stays either way.
|
||||
const quiet = clean(item.acknowledgedReason);
|
||||
return {
|
||||
tone: 'warn',
|
||||
title: message || 'waiting for your reply',
|
||||
tone: approvalTone(item),
|
||||
title: quiet ? `quiet, ${quiet}` : message || 'waiting for your reply',
|
||||
detail: [],
|
||||
options: [],
|
||||
hint: 'p to reply',
|
||||
|
||||
+18
-2
@@ -74,13 +74,25 @@ export function isLiveRow(session: TuiSessionRow): boolean {
|
||||
* outranks a stale `busy` status because the hook is the newer signal. An
|
||||
* errored session has no state of its own here and joins the waiting tier,
|
||||
* since it is equally something only a human can clear.
|
||||
*
|
||||
* ⚠️ An ACKNOWLEDGED item no longer decides the row. `acknowledgedAt` means the
|
||||
* alert this prompt armed has been spent, either because somebody opened the
|
||||
* session on another device or because the inbox opened the item that way for a
|
||||
* session watching its own background work. The item itself stays pending and
|
||||
* answerable, so the row keeps carrying it and the approval card still renders;
|
||||
* it simply stops dragging the session into NEEDS YOU. The web has honoured
|
||||
* that since acknowledgement existed (`approvals-ui.js` clears the pending hook
|
||||
* that `_mobileOverviewState` reads), and this gate is where the TUI had been
|
||||
* reading past it: acknowledging on a phone cleared the alert everywhere except
|
||||
* here. Only `idle` can be acknowledged, so a permission or question dialog is
|
||||
* unaffected by construction, and both are checked ahead of the flag anyway.
|
||||
*/
|
||||
export function classifySession(session: TuiSessionRow, approval?: ApprovalItem): TuiSessionState {
|
||||
if (!isLiveRow(session)) return 'recent';
|
||||
if (approval) {
|
||||
if (approval.kind === 'permission') return 'blocked-permission';
|
||||
if (approval.kind === 'question') return 'blocked-question';
|
||||
return 'waiting';
|
||||
if (!approval.acknowledgedAt) return 'waiting';
|
||||
}
|
||||
if (session.status === 'error') return 'waiting';
|
||||
if (session.isWorking === true || session.status === 'busy') return 'working';
|
||||
@@ -96,7 +108,11 @@ export function classifySession(session: TuiSessionRow, approval?: ApprovalItem)
|
||||
* turn's own start is the pane's last Enter.
|
||||
*/
|
||||
export function stateSince(state: TuiSessionState, session: TuiSessionRow, approval?: ApprovalItem): number {
|
||||
if (approval) return approval.createdAt;
|
||||
// The prompt's own age measures the state only while the prompt is what put the
|
||||
// row in that state. An acknowledged item still rides along on a row that is
|
||||
// plainly idle or working, and dating such a row from it would report how long
|
||||
// ago the prompt arrived as though it were how long the session has been quiet.
|
||||
if (approval && STATE_GROUP[state] === 'needs-you') return approval.createdAt;
|
||||
if (state === 'working') return session.lastSubmitAt ?? session.createdAt ?? 0;
|
||||
return session.lastActivityAt ?? session.createdAt ?? 0;
|
||||
}
|
||||
|
||||
+15
-5
@@ -462,7 +462,8 @@ export function computeListWindow(
|
||||
* The pending dialog, drawn above the tail: the question, the parsed options
|
||||
* with their digits, and the keys that answer them. Red for a permission or
|
||||
* question prompt, yellow for an idle one, the same severity vocabulary the web
|
||||
* inbox uses.
|
||||
* inbox uses. An idle prompt that opened acknowledged carries neither: it reads
|
||||
* grey with the idle glyph, because nothing about it wants the reader.
|
||||
*/
|
||||
export function renderApprovalCard(
|
||||
item: ApprovalItem,
|
||||
@@ -472,8 +473,8 @@ export function renderApprovalCard(
|
||||
): string[] {
|
||||
const paint = painterFor(opts.color);
|
||||
const card = approvalCard(item);
|
||||
const color = card.tone === 'err' ? SGR.red : SGR.yellow;
|
||||
const glyph = card.tone === 'err' ? glyphs.blockedPermission : glyphs.waiting;
|
||||
const color = card.tone === 'err' ? SGR.red : card.tone === 'warn' ? SGR.yellow : SGR.gray;
|
||||
const glyph = card.tone === 'err' ? glyphs.blockedPermission : card.tone === 'warn' ? glyphs.waiting : glyphs.idle;
|
||||
const lines: string[] = [];
|
||||
const push = (text: string, style: string): void => {
|
||||
lines.push(padDisplay(paint(clipStyledLine(text, width), style), width));
|
||||
@@ -571,10 +572,19 @@ function previewBody(
|
||||
// Chrome
|
||||
// ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
/** Sessions with a prompt waiting on a human, which is what the badge counts. */
|
||||
/**
|
||||
* Sessions with a prompt waiting on a human, which is what the badge counts.
|
||||
*
|
||||
* An ACKNOWLEDGED item is not one of them. Its alert has been spent, either by somebody
|
||||
* opening the session elsewhere or because the inbox opened it that way for a session
|
||||
* watching its own background work, and the row has already left NEEDS YOU by the same
|
||||
* flag (`classifySession`). Counting it here would put a number in the header for a
|
||||
* group the reader can see is empty.
|
||||
*/
|
||||
export function pendingApprovalCount(model: TuiRenderModel): number {
|
||||
let count = 0;
|
||||
for (const group of model.groups()) for (const row of group.rows) if (row.approval) count++;
|
||||
for (const group of model.groups())
|
||||
for (const row of group.rows) if (row.approval && !row.approval.acknowledgedAt) count++;
|
||||
return count;
|
||||
}
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user