mirror of
https://github.com/Ark0N/Codeman.git
synced 2026-09-30 12:39:42 +02:00
Compare commits
114
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
72d437ab63 | ||
|
|
1ba0684438 | ||
|
|
af744bdb54 | ||
|
|
51b4a1b758 | ||
|
|
4205f6930f | ||
|
|
12de3c5164 | ||
|
|
9af12afb57 | ||
|
|
fe3bd0074c | ||
|
|
3b55957d79 | ||
|
|
1a99b5836c | ||
|
|
035bfbc2fe | ||
|
|
4c705094f7 | ||
|
|
c376534a50 | ||
|
|
2c3ccdf030 | ||
|
|
475436242c | ||
|
|
613b774bf1 | ||
|
|
60c9af0599 | ||
|
|
2d842ded35 | ||
|
|
5fc391a47c | ||
|
|
a0298cf2b1 | ||
|
|
5bb489addb | ||
|
|
19ffe9b7a8 | ||
|
|
95dc6fe944 | ||
|
|
383f834704 | ||
|
|
e0d4477edc | ||
|
|
5cfb98fb8b | ||
|
|
3edf9aae2f | ||
|
|
358aef16e3 | ||
|
|
1040f6c489 | ||
|
|
e271a65e79 | ||
|
|
3cdb4bf42e | ||
|
|
492f8d8ddf | ||
|
|
56209e7829 | ||
|
|
afb6754453 | ||
|
|
f9edb33d15 | ||
|
|
3730bc7df5 | ||
|
|
cfd771d1d8 | ||
|
|
75a028e825 | ||
|
|
9982a1325f | ||
|
|
1f32128ca9 | ||
|
|
20fc7b3c3d | ||
|
|
0e1191b774 | ||
|
|
bb8ada7e5f | ||
|
|
ea5323d990 | ||
|
|
dee674d3e2 | ||
|
|
f32c4f60d5 | ||
|
|
9a503872d9 | ||
|
|
9d7b29d899 | ||
|
|
ff8dc92187 | ||
|
|
1f61d21298 | ||
|
|
62ceb4e87b | ||
|
|
5108a24bf0 | ||
|
|
5f55f9cb65 | ||
|
|
18ab2ab595 | ||
|
|
e2034177c5 | ||
|
|
8520925e76 | ||
|
|
db9729e1fc | ||
|
|
2d3fc65758 | ||
|
|
5ddc028a2f | ||
|
|
470f75b08c | ||
|
|
211b872335 | ||
|
|
2c89359d42 | ||
|
|
962029bb3d | ||
|
|
b45a96358e | ||
|
|
29984c639d | ||
|
|
acb8d4b0aa | ||
|
|
7b947fa3f1 | ||
|
|
39976041e0 | ||
|
|
71ed7b127c | ||
|
|
fa52753e8b | ||
|
|
993710263d | ||
|
|
7bbe408e44 | ||
|
|
55dae31530 | ||
|
|
0af233c96c | ||
|
|
fbede5cd2a | ||
|
|
a7f74f374f | ||
|
|
0929694012 | ||
|
|
01b32ee6cd | ||
|
|
2936ba6e3d | ||
|
|
83033b4299 | ||
|
|
f865f74a0f | ||
|
|
fbee1b2d82 | ||
|
|
bcebc81fcd | ||
|
|
da933d70be | ||
|
|
25f22b9839 | ||
|
|
97464bfa27 | ||
|
|
0e8b1981af | ||
|
|
5c25a52f95 | ||
|
|
409a6e65f9 | ||
|
|
a1c35da0d8 | ||
|
|
9a9e542a7d | ||
|
|
5a9ff07f57 | ||
|
|
60e1bd52f7 | ||
|
|
fed6582d3e | ||
|
|
98d26e14d9 | ||
|
|
25fae9ad10 | ||
|
|
4a30f510e6 | ||
|
|
d0a5a583cd | ||
|
|
8dfc965d13 | ||
|
|
bd286bf502 | ||
|
|
3248f35081 | ||
|
|
c9515b1d4c | ||
|
|
5b920cb43d | ||
|
|
c4322513d9 | ||
|
|
1380b023e2 | ||
|
|
e8f7772320 | ||
|
|
2f61be6e74 | ||
|
|
8b5a13435a | ||
|
|
3f0bfde54a | ||
|
|
0f3eea2fb5 | ||
|
|
a81f430e41 | ||
|
|
3f2928ae73 | ||
|
|
018f0c4160 | ||
|
|
268e4819ff |
@@ -0,0 +1,5 @@
|
||||
---
|
||||
'aicodeman': minor
|
||||
---
|
||||
|
||||
Installer v2. `curl -fsSL https://getcodeman.com/install | bash` now looks at the machine first, asks at most three questions up front (how the dashboard is reached, optionally what to call the machine on your tailnet, whether to run Codeman as a background service), does the install unattended behind progress spinners with the output in `~/.codeman/install.log`, and ends on the URL with a QR code to scan. One consent covers every missing package and sudo asks for your password once. Flags pipe through `bash -s --` (`--tailscale | --lan | --local`, `--name <n> | --no-rename`, `--service | --run | --no-start`, `--yes`, `--password`, `--port`), `install.sh status` prints the URL and the QR code again, and the cloudflared question moved out of the main flow into `install.sh cloudflared`. On the Tailscale route, a `:443` that already belongs to another app gets Codeman under `https://<node>/codeman` (or on a second port) instead of a dead end, the node can be renamed opt-in (`--name`, `install.sh name`, undone by uninstall), and the HTTPS-certificates toggle is polled with the admin page opened for you. Also fixed on the way: the installer's own `npm install` no longer lets the postinstall start a stray server on port 3000 (the service crash-looped on EADDRINUSE while the done screen said "running"), the LAN address comes from the default route rather than the first interface, a hand-written LaunchDaemon on a headless Mac is left alone, a flag re-run keeps an existing dashboard password, and the done screen's start command carries the sub-path and port it was installed with.
|
||||
@@ -1,5 +0,0 @@
|
||||
---
|
||||
"aicodeman": patch
|
||||
---
|
||||
|
||||
Keep the terminal anchored where you are reading while an agent streams (#358). Scrolling up during a Codex response could still be dragged back to the live bottom by the next redraw: the flush captured the viewport before writing and restored it immediately after, but xterm parses asynchronously, so at that moment the buffer had not moved yet, the restore compared the anchor against itself and did nothing, and the redraw landed a tick later with nothing left to pull the view back. The restore now runs inside xterm's own write callback, which is the first point at which the redraw's effect exists, and it holds across consecutive and chunked redraws. It is dropped if you switch sessions or a history replay starts before the write parses, since the anchor indexes the buffer it was captured from.
|
||||
@@ -10,7 +10,7 @@
|
||||
"name": "codeman",
|
||||
"source": "./plugins/codeman",
|
||||
"description": "Drive Codeman from inside a Claude Code session: spawn worker sessions, prompt them, wait for them, read their answers, clean up. Acts only inside a Codeman-managed session.",
|
||||
"version": "1.29.0",
|
||||
"version": "1.31.0",
|
||||
"author": {
|
||||
"name": "Ark0N",
|
||||
"url": "https://github.com/Ark0N"
|
||||
|
||||
@@ -100,6 +100,50 @@ jobs:
|
||||
fi
|
||||
echo "bash $BASH_VERSION: dsh identity probe survives a missing timeout"
|
||||
'
|
||||
# Installer v2: the question phase runs before the build, and every decision it
|
||||
# takes is bash logic over stubbed tailscale state. Drive the flags, the launch
|
||||
# default, the occupied-:443 menu and the rename question with canned answers,
|
||||
# so a bash-4 construct or a flipped default in any of them fails here, not on a
|
||||
# Mac. The JSON parsers need node (absent in this image) and are stubbed; their
|
||||
# own coverage is test/install-sh-invariants.test.ts plus the vitest gate.
|
||||
docker run --rm -v "$PWD":/w -w /w -e CODEMAN_INSTALL_SH_LIB=1 -e HOME=/tmp/h bash:3.2 bash -c '
|
||||
set -euo pipefail
|
||||
mkdir -p /tmp/h
|
||||
. /w/install.sh
|
||||
parse_flags --tailscale --service --name Build-Box --port 4000
|
||||
[[ "$CODEMAN_TAILSCALE" == "1" && "$LAUNCH_PRESET" == "2" && "$TS_NAME" == "Build-Box" && "$CODEMAN_PORT" == "4000" ]]
|
||||
[[ "$(ts_sanitize_name "$TS_NAME")" == "build-box" ]]
|
||||
has_tty() { return 0; }
|
||||
ANSWER=""; read_reply() { eval "$1=\"\$ANSWER\""; }
|
||||
systemctl() { return 0; }
|
||||
LAUNCH_PRESET=""; NONINTERACTIVE=0
|
||||
choose_launch_mode linux >/dev/null 2>&1
|
||||
[[ "$LAUNCH_CHOICE" == "2" ]]
|
||||
check_tailscale() { return 0; }
|
||||
ts_status_field() { case "$1" in "s.BackendState") printf Running ;; "s.Self && s.Self.DNSName") printf "box.tail.ts.net." ;; esac; }
|
||||
ts_backend_state() { printf Running; }
|
||||
ts_dns_name() { printf box.tail.ts.net; }
|
||||
ts_serve_443_target_port() { printf 8080; }
|
||||
ts_serve_find_port_mapping() { :; }
|
||||
ts_serve_port_used() { return 1; }
|
||||
detect_tailscale_serve_url() { :; }
|
||||
tailscale_choose_mapping >/dev/null 2>&1
|
||||
[[ "$TS_SERVE_MODE" == "path" && "$BIND_BASE_URL" == "/codeman" ]]
|
||||
RENAMED=""; tailscale_rename_node() { RENAMED="$1"; }
|
||||
TS_NAME=""; tailscale_choose_name >/dev/null 2>&1
|
||||
[[ -z "$RENAMED" ]]
|
||||
# A flag re-run keeps the password the unit already carries (and so
|
||||
# never writes the unauthenticated ack), and the hand-start line the
|
||||
# done screen prints carries every non-default value.
|
||||
read_existing_binding() { EXISTING_FOUND=1; EXISTING_HOST=0.0.0.0; EXISTING_PASSWORD=s3cret; EXISTING_ACK=0; EXISTING_BASE_URL=""; }
|
||||
CODEMAN_HOST=0.0.0.0; CODEMAN_TAILSCALE=0; unset CODEMAN_PASSWORD; BIND_ACK=0
|
||||
choose_network_binding >/dev/null 2>&1
|
||||
[[ "$BIND_PASSWORD" == "s3cret" && "$BIND_ACK" == "0" ]]
|
||||
BIND_HOST=0.0.0.0; BIND_PASSWORD=x; BIND_ACK=0; BIND_BASE_URL=/codeman; CODEMAN_PORT=4000
|
||||
[[ "$(start_command_hint)" == "CODEMAN_HOST=0.0.0.0 CODEMAN_PASSWORD="*" CODEMAN_BASE_URL=/codeman CODEMAN_PORT=4000 codeman web" ]]
|
||||
RECONFIGURE=0; parse_flags --port 4001; [[ "$RECONFIGURE" == "1" ]]
|
||||
echo "bash $BASH_VERSION: question phase (flags, launch default, occupied :443, rename opt-in, kept password, start line) ok"
|
||||
'
|
||||
|
||||
- name: CLI catalogue artifacts are in sync with stock.ts
|
||||
run: npm run generate:cli-catalog -- --check
|
||||
|
||||
+137
@@ -1,5 +1,142 @@
|
||||
# aicodeman
|
||||
|
||||
## 1.31.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- 035bfbc: feat(remote): wake a sleeping remote host from Codeman
|
||||
|
||||
A remote SSH case pointing at a machine that suspends used to fail the same way every
|
||||
time: the session was there, the host was not, and typing into it went nowhere. A host
|
||||
can now carry a wake target, either a MAC address for Wake-on-LAN (Codeman builds the
|
||||
magic packet itself, so nothing reaches a shell) or a wake command of your own, and
|
||||
Codeman uses it when you ask for the host: when you type into a sleeping session, when
|
||||
you press the wake button on the banner, or when you start or attach a session on that
|
||||
host. Input you type while it wakes is buffered and flushed once it is back, up to 4 KB,
|
||||
and a chunk over that is refused outright rather than delivered as a fragment.
|
||||
|
||||
Waking only ever happens because you asked. No watcher, dropped-session handler or
|
||||
boot-recovery path can reach it, since a machine woken by a reconnect watcher would come
|
||||
back seconds after every suspend.
|
||||
|
||||
- fbee1b2: feat(custom-model): pick a custom endpoint straight from the Run menu
|
||||
|
||||
#393 landed the backend for custom model endpoints and left it reachable only over the
|
||||
HTTP API. This is the rest of it. Turn on Custom model endpoints in App Settings, save
|
||||
an endpoint, and the Run dropdown grows a Custom Endpoints section built live off the
|
||||
CLI registry, one entry per harness that can actually redirect plus each endpoint you
|
||||
saved. Pick one and it launches that harness pointed at your server, asking which model
|
||||
first when the endpoint has more than one. Endpoints re-discover themselves every five
|
||||
minutes, and one unreachable endpoint never blocks the others. App Settings gains full
|
||||
add, edit and delete for endpoints.
|
||||
|
||||
Seven of the harnesses (opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP) now launch
|
||||
directly onto the endpoint with no restart at all, where before you watched a native
|
||||
boot followed immediately by a second one. Claude still launches and then restarts in
|
||||
place, which its own resume makes far less jarring.
|
||||
|
||||
Most of this release's work went into things that only show up against a real server,
|
||||
and each was found that way rather than in tests: a freshly launched CLI reporting
|
||||
itself busy for its own startup and getting refused; Claude Code assuming a large
|
||||
context window for a model it does not recognise and silently overflowing a small one;
|
||||
a model whose real context is below what Claude Code's own system prompt costs, which
|
||||
no setting can fix and which now warns before launching into a certain failure; and the
|
||||
big one, llama.cpp running exactly one model at a time, so applying a selection can
|
||||
unload the model another session is using. That last case now asks first, tells you
|
||||
which session it affects, and keeps a "loading model" notice on screen for the whole
|
||||
swap window, so a prompt sent mid-swap reads as loading rather than as an answer from
|
||||
whatever was loaded a moment ago. A background sweep also catches the reverse: your
|
||||
session's model being evicted later by somebody else's ordinary use.
|
||||
|
||||
Two things worth knowing if you drive this over the HTTP API or run multi-user. The two
|
||||
questions an apply can ask (the model's context window is too small, and loading it will
|
||||
unload the model another session is using) are now answered by separate
|
||||
`confirmedContext` and `confirmedSwap` fields rather than one `confirmed`. They shared a
|
||||
flag until now, and since the context check runs first, confirming that one silently
|
||||
agreed to evict another session's model as well. The old `confirmed` still means both.
|
||||
And `CLAUDE_CONFIG_DIR` is now admin-only in multi-user mode: it joined claude's
|
||||
privileged env keys, so a non-granted owner can no longer set it through `envOverrides`,
|
||||
and an already-persisted one is dropped on reboot-restore, which returns that session to
|
||||
the default Claude account rather than the per-client one it was pointed at. Single-user
|
||||
installs are unaffected.
|
||||
|
||||
Remote SSH and Docker sessions are refused for now, since their restart reattaches a
|
||||
durable tmux rather than relaunching the agent.
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- c9515b1: fix(terminal): keep the output a pane capture could not contain. Opening a session, a backpressure refresh, a clear-terminal reload and a full-history re-pull all load the screen from a tmux pane capture, and anything the CLI printed between that capture and the end of the load used to be dropped, so its next partial redraw landed on a frame the terminal had never seen: missing or garbled output right after a tab switch or a refresh, plainest in a shell session. Each load now replays exactly the output that arrived after the capture, through one shared rule for all four paths, and a refresh that restores your scroll position no longer snaps back to the bottom afterwards.
|
||||
- 3edf9aa: fix(terminal): replay a pane capture at the geometry it was taken at
|
||||
|
||||
Opening a session could draw a frame built for a pane bigger than your terminal. A
|
||||
taller pane wrote its overflow rows onto the last line and lost the rows underneath
|
||||
(against a 50-row pane, a 30-row terminal rendered 28 of a 45-line command and drew
|
||||
the survivors twice), and a wider one wrapped every row and scrolled the whole frame
|
||||
up by one. The terminal response now reports the geometry the capture was really
|
||||
taken at, so the browser can see the mismatch and replay once at the size that stuck.
|
||||
A pane that cannot be sized to fit is diagnosed once per session instead of on every
|
||||
tab switch.
|
||||
|
||||
- 035bfbc: ### Thanks
|
||||
- @irisitymichaelgrundberg for three terminal fixes in one release: keeping the output a pane capture could not contain (#436), replaying a capture at the geometry it was taken at (#435, five rounds and a Playwright suite that fails against the merge base), and trimming the padding out of a copied selection (#451), where the scan-instead-of-regex call avoided a 2.9s freeze nobody would have traced back to a copy.
|
||||
- @timkjr for a first contribution that found a real silent failure: the Instance count stepper next to the Run button had only ever applied to Claude, so on the other eight run modes it launched one session and said nothing (#454).
|
||||
- @Randalix for Wake-on-LAN on remote hosts (#439), built and live-tested against a real sleeping machine, and for reading the whole diff again between rounds rather than only the parts that were asked about.
|
||||
- @opticon454 for turning #393's backend-only custom model endpoints into the whole feature (#430), and for validating it against a real llama-swap box rather than against the tests: the `/props` versus `/running` context discrepancy and the DeepSeek `/v1` root cause were both tracked down to the SDK source instead of guessed at.
|
||||
|
||||
- c376534: fix(run): make the Instance count stepper work for every non-Claude mode
|
||||
|
||||
The Instance count stepper next to the Run button only ever applied to Claude.
|
||||
Setting it to 3 and launching OpenCode, Codex, Gemini, Antigravity, Pi, OMP, Grok or
|
||||
DeepSeek started exactly one session, with no error and no hint that the control had
|
||||
done nothing. All eight now launch the count you asked for, and the opening banner
|
||||
says how many are starting. The one exception is a launch started from the Custom
|
||||
Endpoints section of the Run menu, which always starts a single session.
|
||||
|
||||
- 19ffe9b: fix(input): make sure a prompt sent through the API actually leaves the composer. Claude Code 2.1.277 started ignoring Enter for the first 30 to 50 seconds after the composer paints while still accepting the typed text, so a prompt sent right after a session came up sat unsent in the pane and every waiter (send-and-wait, the agent skill, cron, the maintainer bot) burned its whole timeout on a turn that never started. The server now reads the pane after every programmatic write that carried Enter and presses Enter again, on a 2 to 60 second schedule, only while the composer verifiably still holds the text it sent; an empty composer, other text, or a pane with no composer at all ends it. The agent skill's `sendwait` gets the same loop for servers that predate this, and its preamble version moves to 1.30.1 so an already-seeded agent picks up the fresh copy.
|
||||
- f9edb33: fix(terminal): trim the padding out of a copied selection
|
||||
|
||||
Copying out of a pane put a wall of spaces on the clipboard. xterm hands back
|
||||
whole screen rows and trims only the cells that were never written to, so the
|
||||
real spaces a full-screen program paints across the unused part of a row count
|
||||
as content: measured against Claude Code in a 282-column pane, single lines
|
||||
arrived carrying 138 trailing spaces. Pasting that into a chat client or an
|
||||
editor meant deleting the whitespace by hand, while Windows Terminal, iTerm2 and
|
||||
GNOME Terminal all trim it for you. A copy now drops the trailing run from every
|
||||
line, on all four paths (the Ctrl+C chord, right-click, the phone selection
|
||||
button and Auto Copy), while leading indentation is left exactly as it is. An
|
||||
Alt+drag rectangular selection is copied verbatim, because its columns lining up
|
||||
is the point of that gesture. A selection holding nothing but padding is refused
|
||||
rather than copied as bare line breaks.
|
||||
|
||||
## 1.30.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- da933d7: Offer to rebuild the sessions a host reboot destroyed. A reboot takes the tmux server down with it, so every pane dies and the board comes up empty. Codeman now works out what was running, and the board offers to restore it behind a click. The conversations come back; the terminal scrollback does not, and the banner says so.
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- a1c35da: Stop a phone keyboard losing the last character of every message it sends. Android soft keyboards commit the last typed character and send the Enter key in one InputConnection transaction, so the `input` event and the Enter keydown are both processed before any zero-delay timer runs. The orphaned-input recovery from #388 only resolved its candidate on such a timer, and lost it both ways: xterm emits `\r` synchronously from the Enter keydown, so the local-echo composer submitted the prompt before the recovered character existed, and that `\r` bumped the "did xterm speak for this keystroke" counter, so the candidate then stood itself down and dropped the character outright. Pending candidates are now drained synchronously at the next keydown, from xterm's custom key handler, which runs before xterm processes that key, so the counter still holds the value it had while the candidate's own keystroke was current, and the recovered byte reaches the composer ahead of the Enter. Typing on a physical keyboard is unaffected: there, the timer has already resolved the candidate before the next key arrives.
|
||||
- 3f2928a: The installer's hint for a launcher-only CLI (DeepSeek today) now says why it is a docs link rather than a command you can run, and points at the thing that resolves it: the package installs a launcher that still needs a terminal profile, and Codeman's Run menu can add one in a click. Driven by a generated `CLI_LAUNCHER_ONLY` flag rather than an id check, so it covers any future entry of that shape. Also removes three dead lookup helpers and two never-read generated arrays from `install.sh`, skips a disabled entry's probe instead of filtering it afterwards, and corrects a comment that claimed the non-interactive default is always Claude Code (on a wget-only host its curl one-liner is filtered out first).
|
||||
- 0e1191b: Maintainer fixes applied while landing the above. A session restored after a reboot keeps the name you gave it (the rebuild dropped the field that records who named a session, so a hand-renamed session came back looking auto-named and the next prompt overwrote it), and no longer types `continue` into itself on its own: a pending auto-resume stamp from before the reboot is dropped rather than re-armed, since the pane is new and one click could otherwise arm several unattended prompts at once. Auto-resume itself stays on and re-arms on the next real usage-limit message. The restore offer is also hidden in a detached single-session window, which has no tab strip to put restored sessions in, and a conversation that goes live while an earlier session in the same batch is starting is no longer restored a second time.
|
||||
- 0e1191b: ### Thanks
|
||||
- @irisitymichaelgrundberg for the reboot-restore banner (#442), and for the three real reboots behind it rather than a mocked one.
|
||||
- @shenlvkang-collab for tracking down why Android keyboards lost the last character of every message (#441), including the half where the character was not late but gone.
|
||||
- @opticon454 for going back and closing out the loose ends left as "worth knowing rather than fixing" after #380 (#429).
|
||||
|
||||
- de864e7: Keep the terminal anchored where you are reading while an agent streams (#358). Scrolling up during a Codex response could still be dragged back to the live bottom by the next redraw: the flush captured the viewport before writing and restored it immediately after, but xterm parses asynchronously, so at that moment the buffer had not moved yet, the restore compared the anchor against itself and did nothing, and the redraw landed a tick later with nothing left to pull the view back. The restore now runs inside xterm's own write callback, which is the first point at which the redraw's effect exists, and it holds across consecutive and chunked redraws. It is dropped if you switch sessions or a history replay starts before the write parses, since the anchor indexes the buffer it was captured from.
|
||||
|
||||
## 1.29.1
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- 5b920cb: Auto-name sessions from the first prompt (#376, opt-in). With the new synced **Auto-name Sessions** setting on (App Settings → Appearance → Tabs, default off), a tab that still carries its generated name takes a title from the first real prompt you submit, keeping the case prefix: `w3-myapp` becomes `w3-myapp: fix the login redirect`. The strip shows the title with the prefix in the tooltip, and the next session in that case still counts up. It happens once per session, only for prompts you type or send through the input API (never a Ralph, respawn, cron or approval answer), never for shells, and a name you set yourself is never touched. Slash commands such as `/clear` do not become titles. The title is derived locally from the prompt's first sentence; no text leaves the machine. `nameSource` (`placeholder` / `auto` / `manual`) is a new additive field on session state.
|
||||
|
||||
Landed with the fixes the review of #376 asked for: first prompt only (not every prompt), a user-input gate so Ralph, respawn, cron and approval writes cannot name a tab, the prefix form so the case identity and `w<n>` counter survive, and a keystroke tracker that handles a bare Esc, bracketed pastes, wheel reports, Tab and history recall instead of mis-titling the tab.
|
||||
|
||||
### Thanks
|
||||
- @shenlvkang-collab for #376, the auto-naming idea and the ownership plumbing (`nameSource`, the listener wiring, the restore path) it shipped with.
|
||||
|
||||
## 1.29.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
<h2 align="center">Mission control for AI coding agents</h2>
|
||||
|
||||
<p align="center">
|
||||
<em>Claude Code • OpenCode • Codex • Antigravity • Gemini • Pi • Grok • OMP • Terminal - One Dashboard • Any Device</em>
|
||||
<em>Claude Code • OpenCode • Codex • Antigravity • Gemini • Pi • Grok • DeepSeek • OMP • Terminal - One Dashboard • Any Device</em>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
@@ -27,7 +27,7 @@
|
||||
<img src="docs/images/subagent-demo-20260724.gif" alt="Codeman — parallel subagent visualization" width="900">
|
||||
</p>
|
||||
|
||||
**Codeman** is a self-hosted mission control for AI coding agents. It spawns Claude Code, OpenCode, Codex, Antigravity, Gemini, Pi, Grok, or OMP inside persistent tmux sessions, streams the real terminal to any browser, and keeps agents productive after you walk away: it re-prompts on idle, resumes when a usage limit resets, runs scheduled jobs, and shows every background agent working in real time.
|
||||
**Codeman** is a self-hosted mission control for AI coding agents. It spawns Claude Code, OpenCode, Codex, Antigravity, Gemini, Pi, Grok, DeepSeek Harness, or OMP inside persistent tmux sessions, streams the real terminal to any browser, and keeps agents productive after you walk away: it re-prompts on idle, resumes when a usage limit resets, runs scheduled jobs, and shows every background agent working in real time.
|
||||
|
||||
Get started in one line (macOS & Linux, Windows via WSL):
|
||||
|
||||
@@ -42,7 +42,7 @@ codeman web
|
||||
|
||||
The installer asks before every system change, and re-running the same line updates in place. Full details: [Quick Start - Installation](#quick-start---installation).
|
||||
|
||||
- **One dashboard, eight CLIs** - run [Claude Code, OpenCode, Codex, Antigravity, Gemini, Pi, Grok, or OMP](#more-features) per session (plus plain shell), locally, [in Docker](#isolated-docker-sessions), or [over SSH](#remote-ssh-sessions)
|
||||
- **One dashboard, nine CLIs** - run [Claude Code, OpenCode, Codex, Antigravity, Gemini, Pi, Grok, DeepSeek, or OMP](#more-features) per session (plus plain shell), locally, [in Docker](#isolated-docker-sessions), or [over SSH](#remote-ssh-sessions), with your own dashboards open as [web tabs](#more-features) beside them
|
||||
- **Truly phone-friendly** - a [touch-optimized terminal](#mobile-optimized-web-ui) with instant local echo, QR login, swipe navigation, and push notifications
|
||||
- **Runs while you sleep** - [idle detection + respawn cycling](#respawn-controller) and auto-resume when a subscription limit resets, for 24+ hour unattended runs
|
||||
- **See your agents think** - [live floating windows](#live-agent-visualization) for every subagent and teammate, with real-time transcripts
|
||||
@@ -61,14 +61,15 @@ The installer asks before every system change, and re-running the same line upda
|
||||
curl -fsSL https://getcodeman.com/install | bash
|
||||
```
|
||||
|
||||
This installs Node.js, tmux and a build toolchain if missing (node-pty ships no Linux prebuilds, so it compiles from source), clones Codeman to `~/.codeman/app`, and builds it. A few things worth knowing:
|
||||
This installs Node.js, tmux and a build toolchain if missing (node-pty ships no Linux prebuilds, so it compiles from source), clones Codeman to `~/.codeman/app`, and builds it. It looks at what is already on the machine, asks at most three questions, then does all the work unattended and ends on the URL with a QR code for your phone. A few things worth knowing:
|
||||
|
||||
- **It asks first.** Every system change (package installs, AI CLI download) is prompted, and a menu at the end lets you choose: run Codeman in this terminal, install it as a background service (systemd/launchd, auto-start on boot), or don't start yet. Nothing runs in the background unless you pick it.
|
||||
- **How it's reachable, your choice.** The installer offers three ways to reach the dashboard: **Tailscale** (loopback bind fronted by `tailscale serve`, so you get `https://<machine>.<tailnet>.ts.net` with a real certificate and your tailnet as the login, no password needed), **any device on your network** (`0.0.0.0`, with a strongly recommended password prompt), or **this machine only** (`127.0.0.1`, safest). Skipping the password on a network bind requires an explicit confirmation and ends with a loud warning. The highlighted default reflects what is already on the machine (Tailscale when it is already in use, your existing binding on a re-run), and a bare Enter never pulls in new software. A bare `codeman web` started by hand still defaults to loopback.
|
||||
- **Re-run to update.** The same one-liner updates a finished install in place: local changes in `~/.codeman/app` are stashed (never discarded), and a running service is restarted and verified. If a first install was interrupted, re-running resumes the full setup instead. `install.sh update` and `install.sh uninstall` also exist.
|
||||
- **CI / headless:** without a terminal attached, steps that would change your system abort with instructions instead of running silently. Set `CODEMAN_NONINTERACTIVE=1` to approve them for automation.
|
||||
- **Three questions, all up front.** How the dashboard is reached, optionally what to call this machine on your tailnet, and whether to run Codeman as a background service (systemd/launchd, auto-start on boot; Enter says yes). Everything that needs you, including one consent for all missing packages, one sudo password, and the Tailscale login, happens before the build, so you can walk away while it compiles.
|
||||
- **How it's reachable, your choice.** **Tailscale** (loopback bind fronted by `tailscale serve`, so you get `https://<machine>.<tailnet>.ts.net` with a real certificate and your tailnet as the login, no password needed), **any device on your network** (`0.0.0.0`, with a strongly recommended password prompt), or **this machine only** (`127.0.0.1`, safest). Skipping the password on a network bind requires an explicit confirmation and ends with a loud warning. The highlighted default reflects what is already on the machine (Tailscale when it is already connected, your existing binding on a re-run), and a bare Enter never pulls in new software. If another app already owns `:443` on your node, Codeman goes under `https://<machine>.<tailnet>.ts.net/codeman` or on a second port instead of replacing it. A bare `codeman web` started by hand still defaults to loopback.
|
||||
- **The name is yours to choose.** By default the URL uses the machine's existing tailnet name. Answering yes to the second question renames the machine to `codeman-<hostname>` (which also renames it for SSH, so the default is no); `install.sh name` does it later.
|
||||
- **Re-run to update.** The same one-liner updates a finished install in place: local changes in `~/.codeman/app` are stashed (never discarded), and a running service is restarted and verified. If a first install was interrupted, re-running resumes the full setup instead. `install.sh status` prints the URLs and the QR code again; `install.sh update`, `install.sh tailscale` and `install.sh uninstall` also exist.
|
||||
- **Flags for the impatient.** `curl -fsSL https://getcodeman.com/install | bash -s -- --tailscale --service` answers the questions from the command line (`--lan`, `--local`, `--run`, `--no-start`, `--name <n>`, `--port <n>`, `--yes` too). **CI / headless:** without a terminal attached, steps that would change your system abort with instructions instead of running silently; set `CODEMAN_NONINTERACTIVE=1` to approve them for automation.
|
||||
|
||||
You'll need at least one AI coding CLI installed — [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), [Gemini CLI](https://github.com/google-gemini/gemini-cli), [Pi](https://pi.dev), [Grok Build](https://github.com/xai-org/grok-build), [DeepSeek Harness](https://github.com/deepseek-ai/deepseek-harness), or [OMP](https://github.com/can1357/oh-my-pi) (any combination works; Gemini CLI is enterprise-only since Google's consumer cutover, and Antigravity is its successor). The installer detects whichever of the nine is present; if none is found, it offers to install Claude Code or OpenCode, or you can skip and install one yourself later. After install:
|
||||
You'll need at least one AI coding CLI installed — [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), [Gemini CLI](https://github.com/google-gemini/gemini-cli), [Pi](https://pi.dev), [Grok Build](https://github.com/xai-org/grok-build), [DeepSeek Harness](https://github.com/deepseek-ai/deepseek-harness), or [OMP](https://github.com/can1357/oh-my-pi) (any combination works; Gemini CLI is enterprise-only since Google's consumer cutover, and Antigravity is its successor). The installer detects whichever of the nine is present; if none is found, it offers to install any of them from a menu (DeepSeek excepted, since its npm package installs only a launcher with no runnable profile), or you can skip and install one yourself later. After install:
|
||||
|
||||
```bash
|
||||
codeman web
|
||||
@@ -82,7 +83,7 @@ codeman users add alice --admin # create the first admin account
|
||||
codeman web --multiuser # named logins + per-user case spaces
|
||||
```
|
||||
|
||||
**Prefer Docker Compose?** A local-image Compose deployment ships in `docker/`: copy `docker/.env.example` to `docker/.env`, set `CODEMAN_PASSWORD`, then run `bash docker/Start-Codeman.sh` on Linux. Codeman runs in a container and spawns Docker cases as sibling containers through the host socket. See the [Docker deployment guide](docker/README.md) for direct Compose commands, storage and networking options.
|
||||
**Prefer Docker Compose?** A local-image Compose deployment ships in `docker/`: copy `docker/.env.example` to `docker/.env`, set `CODEMAN_PASSWORD`, then run `bash docker/Start-Codeman.sh` on Linux. Codeman runs in a container and spawns Docker cases as sibling containers through the host socket. After updating, run the script again rather than a plain `docker compose up`, so the rebuilt image, refreshed volumes and entrypoint arrive together. See the [Docker deployment guide](docker/README.md) for direct Compose commands, storage and networking options.
|
||||
|
||||
Details in [Multi-User Mode](#multi-user-mode-opt-in) below.
|
||||
|
||||
@@ -212,14 +213,14 @@ The most responsive AI coding agent experience on any phone. Full xterm.js termi
|
||||
- **Keyboard accessory bar** — `/init`, `/clear`, `/compact` quick-action buttons above the virtual keyboard; destructive commands require a double-press to confirm, so you never fire one by accident; on Codex sessions the bar also shows `⇧←` / `⇧→` (Shift+Left / Shift+Right: edit the last queued message / return through the prompt stack)
|
||||
- **Dedicated Enter button** — replays the keypress through the terminal, so text buffered by local echo is flushed first rather than stranded
|
||||
- **Swipe navigation & smart keyboard handling** — swipe left/right to switch sessions; toolbar and terminal shift up when the keyboard opens (`visualViewport` API)
|
||||
- **Built for phones** — safe-area insets for notch and home indicator, 44px touch targets, bottom-sheet case picker, native momentum scrolling
|
||||
- **Built for phones** — safe-area insets for notch and home indicator, 44px touch targets, bottom-sheet case picker, native momentum scrolling; on a folding phone (iPhone Duo) dialogs stay clear of the hinge, and opening or closing the device is never mistaken for the keyboard
|
||||
|
||||
```bash
|
||||
codeman web --https
|
||||
# Open on your phone: https://<your-ip>:3000
|
||||
```
|
||||
|
||||
> `localhost` works over plain HTTP. Use `--https` when accessing from another device, or use [Tailscale](https://tailscale.com/) (recommended): the installer can set it up for you (choose **Tailscale** at the network-access prompt, or run `bash ~/.codeman/app/install.sh tailscale` on an existing install). That gives you `https://<your-machine>.<tailnet>.ts.net` with a real certificate: private to your tailnet, no password required, and PWA install + push notifications work on your phone.
|
||||
> `localhost` works over plain HTTP. Use `--https` when accessing from another device, or use [Tailscale](https://tailscale.com/) (recommended): the installer can set it up for you (choose **Tailscale** at the network-access prompt, or run `bash ~/.codeman/app/install.sh tailscale` on an existing install). That gives you `https://<your-machine>.<tailnet>.ts.net` with a real certificate: private to your tailnet, no password required, and PWA install + push notifications work on your phone. The installer ends on that URL with a QR code to scan, and `bash ~/.codeman/app/install.sh status` prints it again any time.
|
||||
|
||||
### Secure QR Code Authentication
|
||||
|
||||
@@ -255,7 +256,7 @@ Click **+ New Session** (or **Quick Start**). A session is one AI CLI running in
|
||||
| Field | What it does |
|
||||
| ---------------------------- | ------------------------------------------------------------------------------------------------------------------- |
|
||||
| **Working directory / case** | The folder the agent operates in. A "case" is just a named working dir Codeman remembers. **Add Case** creates one from scratch, links an existing folder, or clones a GitHub repo straight into one (**Clone Repo**). |
|
||||
| **CLI / run mode** | `Claude` (default), `OpenCode`, `Codex`, `Antigravity`, `Gemini`, `Pi`, `Grok`, `OMP`, or `Terminal` (plain shell). |
|
||||
| **CLI / run mode** | `Claude` (default), `OpenCode`, `Codex`, `Antigravity`, `Gemini`, `Pi`, `Grok`, `DeepSeek`, `OMP`, or `Terminal` (plain shell). |
|
||||
| **Model** | Per-session model (App Settings → Models → New Claude sessions). A soft default — `/model` still works in-session. |
|
||||
| **Effort / Ultracode** | Reasoning effort (`low`–`max`) or `ultracode` for dynamic multi-agent workflows. Switchable anytime with `/effort`. |
|
||||
|
||||
@@ -263,7 +264,7 @@ Hit start — Codeman spawns the CLI via a real PTY and streams it to your brows
|
||||
|
||||
### 3. Read the dashboard
|
||||
|
||||
- **Tabs (top)** — one per session. `Alt+1`-`9` to jump, `Ctrl+Tab` for next, drag to reorder (tab order syncs across your devices).
|
||||
- **Tabs (top)** — one per session. `Alt+1`-`9` to jump, `Ctrl+Tab` for next, drag to reorder (tab order syncs across your devices). Prefer a list? **App Settings → Appearance → Tabs** moves it into a left sidebar with a filter box (`Alt+B` collapses it) or a vertical rail whose rows sort by activity: blocked on you first, then longest running, then most recently quiet.
|
||||
- **Terminal (center)** — a real `xterm.js` terminal; full TUIs render correctly. Type directly and press **Enter** to send. `Shift+Enter` inserts a newline.
|
||||
- **Side panels** — Respawn, Orchestrator, Cron, Subagents, Settings (toggled from the toolbar).
|
||||
|
||||
@@ -271,8 +272,10 @@ Hit start — Codeman spawns the CLI via a real PTY and streams it to your brows
|
||||
|
||||
- **Type prompts** straight into the terminal — input is delivered exactly-once even across reconnects (a dropped link never loses or double-sends a prompt).
|
||||
- **Paste or drag-and-drop images** directly into the session.
|
||||
- **Voice input** — `Ctrl+Shift+V` (Deepgram Nova-3, with auto-silence stop).
|
||||
- **Attachments** — register external files/docs and preview Office/PDF inline.
|
||||
- **Voice input** — `Ctrl+Shift+V` (Deepgram Nova-3, or this machine's Claude Code login with no API key; auto-silence stop).
|
||||
- **Attachments** — register external files/docs and preview Office/PDF inline; any file path an agent prints is clickable, in the terminal and in the chat view.
|
||||
- **When it needs you** — the tab turns yellow (waiting for input) or red (a question is blocking). The **Approvals Inbox** _(opt-in)_ queues every pending prompt across sessions, answerable from the header bell or the phone home screen, and 🧠 **Read My Mind** _(opt-in)_ drafts your next prompt from the case's goals and recent work.
|
||||
- **Copy what you see** — `Shift+drag` selects text even while the CLI owns the mouse, right-click copies it, and Auto Copy _(opt-in)_ copies a selection the moment you release it.
|
||||
|
||||
### 5. Make it autonomous
|
||||
|
||||
@@ -291,7 +294,7 @@ Hit start — Codeman spawns the CLI via a real PTY and streams it to your brows
|
||||
|
||||
### 7. Operate & maintain
|
||||
|
||||
- **App Settings** — model, effort, permission startup mode, theme/skin, notifications, display toggles, per-CLI options, a synced custom display name, and per-device English/Simplified Chinese UI language.
|
||||
- **App Settings** — model, effort, permission startup mode, theme/skin, terminal font family and weight, entrance animations, notifications, display toggles, per-CLI options, a synced custom display name, and per-device English/Simplified Chinese UI language.
|
||||
- **Run it in the background** — `codeman web -d` detaches from your shell (`--status`, `--stop`); `codeman service install` makes it a systemd user unit / macOS LaunchAgent that survives reboots. Both verify the server actually answers before reporting success, and both refuse to start a second server on one data dir. See [Keep it running in the background](#quick-start---installation).
|
||||
- **Self-update** — git-clone installs update in place from **App Settings → System → Updates**.
|
||||
- **Deploy your own changes** — see [Development](#development).
|
||||
@@ -439,16 +442,21 @@ PTY Output → 16ms Server Batch → DEC 2026 Wrap → SSE → Client rAF → xt
|
||||
- **Background daemon & service install** — `codeman web -d` runs the server detached with a pidfile, `~/.codeman/web.log`, and verified startup (it polls the server until it answers, so a port clash never reads as success); `codeman service install` writes a systemd user unit (Linux) or LaunchAgent (macOS) with your shell's PATH baked in, so an nvm or Homebrew `node`, `tmux` and `claude` are actually found. Secrets are never written into unit files
|
||||
- **Self-update** — git-clone installs under systemd/launchd update in place from **App Settings → System → Updates**: it detects the latest release, auto-stashes a dirty tree, and streams build progress across the service restart (npm installs report as non-updatable)
|
||||
- **Clone a GitHub repo as a case** — paste a repository URL into **Add Case → Clone Repo** and Codeman clones it into `~/codeman-cases/<name>` and registers it as a normal case, ready to run an agent in. It preflights the URL while you type (tells you whether it can be cloned anonymously and offers the repo's real branches and tags for the optional branch/tag field), fills the case name in from the URL, and lets you pick which CLI the Run button should use. Public repositories over `https://`; Codeman never collects or stores credentials
|
||||
- **Multi-CLI** — run **Claude Code**, **OpenCode**, **Codex**, **Antigravity**, **Gemini**, **Pi**, **Grok**, or **OMP** per session; env-var prefixes auto-gate (`CLAUDE_CODE_*` vs `OPENCODE_*` vs `CODEX_*` vs `ANTIGRAVITY_*` vs `GEMINI_*`/`GOOGLE_*` vs `PI_*` vs `GROK_*`/`XAI_*` vs `OMP_*`). See [`docs/opencode-integration.md`](docs/opencode-integration.md), [`docs/pi-integration.md`](docs/pi-integration.md), [`docs/grok-integration.md`](docs/grok-integration.md) and [`docs/omp-integration.md`](docs/omp-integration.md)
|
||||
- **Docker sessions** — run a case inside an isolated, hardened container. One checkbox on **Create New** spins up a container with sensible defaults and starts the agent inside it; multiple sessions share one per-case container; export a container + its workspace to a portable `.tar.gz` to move it to another machine. See [`docs/docker-cases.md`](docs/docker-cases.md)
|
||||
- **Remote SSH sessions** — point a case at another machine and run the agent there inside a durable remote tmux: survives SSH drops, auto-reconnects, and can discover + attach sessions already running on the host. See [`docs/remote-sessions.md`](docs/remote-sessions.md)
|
||||
- **Multi-CLI** — run **Claude Code**, **OpenCode**, **Codex**, **Antigravity**, **Gemini**, **Pi**, **Grok**, **DeepSeek Harness**, or **OMP** per session; env-var prefixes auto-gate (`CLAUDE_CODE_*` vs `OPENCODE_*` vs `CODEX_*` vs `ANTIGRAVITY_*` vs `GEMINI_*`/`GOOGLE_*` vs `PI_*` vs `GROK_*`/`XAI_*` vs `DSH_*`/`DEEPSEEK_*` vs `OMP_*`). See [`docs/opencode-integration.md`](docs/opencode-integration.md), [`docs/pi-integration.md`](docs/pi-integration.md), [`docs/grok-integration.md`](docs/grok-integration.md), [`docs/deepseek-integration.md`](docs/deepseek-integration.md) and [`docs/omp-integration.md`](docs/omp-integration.md)
|
||||
- **Custom model endpoints** _(new in 1.29.0, HTTP API for now)_ — point a session's CLI at any OpenAI-compatible endpoint instead of its native backend: a local llama.cpp, llama-swap, Ollama or vLLM box, or a cloud gateway such as Azure AI Foundry or OpenRouter. Save an endpoint once (`POST /api/model-endpoints`; its models are discovered from `/v1/models`), apply it to a session (`POST /api/sessions/:id/custom-model`), and the CLI restarts in place on that endpoint. Verified live for Claude, OpenCode, Pi, Grok and OMP; Codex, Gemini and DeepSeek have documented gaps, Antigravity has no mechanism. A toolbar picker is the follow-up. See [`docs/custom-model-endpoints.md`](docs/custom-model-endpoints.md)
|
||||
- **Web tabs** — open Grafana, Uptime Kuma, a Vite dev server or any dashboard URL as a tab beside your sessions (Run dropdown → **Web / URL** → **Add URL**). Dashboards are proxied through Codeman's own origin, so an `http://` target works from a phone over HTTPS and through the tunnel, single-page apps route on their own paths, and a frame that reloads recovers itself. A `localhost` link an agent prints opens as a web tab automatically. See [`docs/web-tabs.md`](docs/web-tabs.md)
|
||||
- **Docker sessions** — run a case inside an isolated, hardened container. One checkbox on **Create New** spins up a container with sensible defaults and starts the agent inside it; multiple sessions share one per-case container, or attach a case to a container you already run; export a container + its workspace to a portable `.tar.gz` to move it to another machine. See [`docs/docker-cases.md`](docs/docker-cases.md)
|
||||
- **Remote SSH sessions** — point a case at another machine and run the agent there inside a durable remote tmux: survives SSH drops, auto-reconnects, and can discover + attach sessions already running on the host; file previews and downloads come over the same ssh connection. See [`docs/remote-sessions.md`](docs/remote-sessions.md)
|
||||
- **Effort & Ultracode** — set a per-session default effort (`low`–`max`) or enable **ultracode** (dynamic multi-agent workflows). Soft defaults only — switchable anytime with `/effort` in-session. Extended-thinking budget is configurable too
|
||||
- **Voice input** — dictate prompts with Deepgram Nova-3 (Web Speech API fallback): toggle recording, auto-silence stop, live level meter (`Ctrl+Shift+V`)
|
||||
- **Voice input** — dictate prompts with Deepgram Nova-3, or through this machine's Claude Code login with no API key at all (App Settings → Voice; Web Speech API fallback): toggle recording, auto-silence stop, live level meter (`Ctrl+Shift+V`)
|
||||
- **Image input** — paste or drag-and-drop images straight into a session
|
||||
- **Gesture control** _(opt-in)_ — a MediaPipe hand-tracking overlay to grab/drag session windows and pinch buttons, hands-free. Enable with `CODEMAN_GESTURE=1` + App Settings → Terminal & Input
|
||||
- **Multi-monitor span** _(macOS)_ — one click opens a browser window maximized across all displays, so floating agent/gesture panels can cross the physical seam
|
||||
- **File Viewer button** _(opt-in)_ — a header button that toggles the built-in file browser panel with one tap; enable under App Settings → Header & Panels → Header buttons
|
||||
- **CJK / IME input** — full composition support for Chinese / Japanese / Korean
|
||||
- **CJK / IME input** — full composition support for Chinese / Japanese / Korean, with Ctrl- and Alt-modified navigation keys passed through to the CLI
|
||||
- **Plan usage in the header** — live Claude subscription usage (the 5-hour and weekly windows) from a statusline exporter Codeman hands to `claude` at spawn and never writes into your settings files, plus Codex limits from its own app-server; per device, on for desktops and off for phones
|
||||
- **Session list, your way** — the header strip, a left sidebar with a filter box, or a vertical rail whose detailed rows carry created and state stamps and sort by activity; the phone home screen and the desktop home rail use the same order
|
||||
- **Terminal looks** — seven skins, four of them light, per-device font family and weight (the bundled JetBrains Mono covers weights 100 to 800), and opt-in entrance animations for tabs, agent windows, the terminal pane and connection lines
|
||||
- **OS notifications & hostname-aware titles** — desktop alerts and tab titles are prefixed `codeman:<host>` so multi-host setups stay unambiguous
|
||||
|
||||
---
|
||||
@@ -461,8 +469,9 @@ Run a case inside its own hardened Docker container instead of directly on your
|
||||
- **Resource templates** — expand the checkbox for a **Small / Medium / Large / GPU** preset (memory, CPUs, GPU), or set your own. **Disk is elastic** — storage grows as data flows in, no fixed cap.
|
||||
- **Shared per-case container** — many sessions can `docker exec` into the same container; killing one session never tears the container out from under the others.
|
||||
- **Hardened by default** — non-root, `--cap-drop ALL`, `no-new-privileges`, PID/memory caps, never `--privileged` or the docker socket; a **sealed** profile (no host credentials, network off) is one toggle away.
|
||||
- **Seamless auth, isolated credentials** — your host Claude / Codex / Antigravity / Gemini / OpenCode / Pi logins work inside the container out of the box: credentials are seeded (copied) in at launch and onboarding/trust prompts are pre-answered, so no login wizard appears. The container keeps its own copies and never writes back to your host credential stores; only conversation transcripts are shared, and exports never capture secrets.
|
||||
- **Seamless auth, isolated credentials** — your host Claude / Codex / Antigravity / Gemini / OpenCode / OMP logins work inside the container out of the box: credentials are seeded (copied) in at launch and onboarding/trust prompts are pre-answered, so no login wizard appears. The container keeps its own copies and never writes back to your host credential stores; only conversation transcripts are shared, and exports never capture secrets.- **Move it to another machine** — export a container's whole environment (toolchain + workspace) to a portable `.tar.gz`, `docker load` it on the other side, and import it into a fresh case.
|
||||
- **Seamless auth, isolated credentials** — your host Claude / Codex / Antigravity / Gemini / OpenCode / Pi / Grok / OMP logins work inside the container out of the box: credentials are seeded (copied) in at launch and onboarding/trust prompts are pre-answered, so no login wizard appears. The container keeps its own copies and never writes back to your host credential stores; only conversation transcripts are shared, and exports never capture secrets.
|
||||
- **Attach to a container you already run** — tick **Attach to an existing container** on the Docker panel to link a case to it instead of creating one. Codeman only `exec`s into it and never starts, stops, restarts or removes it; one adopted container can back several cases at different directories, and **copy an existing case** pre-fills the form from a sibling. Admin-only in multi-user mode, since the container's mounts belong to whoever started it.
|
||||
- **Move it to another machine** — export a container's whole environment (toolchain + workspace) to a portable `.tar.gz`, `docker load` it on the other side, and import it into a fresh case.
|
||||
- **Durable** — reconnect after a restart lands back in the same live agent; a container stop/reboot resumes the conversation from the bind-mounted transcript.
|
||||
|
||||
Prerequisite: just Docker (or Podman). The agent base image builds itself automatically on first use, with progress streamed to the UI (or pre-build it with `node scripts/build-agent-image.mjs`). Full guide: [`docs/docker-cases.md`](docs/docker-cases.md).
|
||||
@@ -478,6 +487,7 @@ Point a case at another machine and run the agent **there**, over SSH, with the
|
||||
- **Discover & attach**: list the `codeman-*` sessions already running on a host (started by that machine's own Codeman, or by another operator) and attach to one. Attached sessions you don't own **detach on tab close, never kill**.
|
||||
- **Shared sessions**: several clients can attach the same remote session at different window sizes without clamping each other; discovery shows a "shared" badge with the client count.
|
||||
- **Injection-safe**: every ssh command line flows through a single shell-escaping builder, and host/path/identity fields are schema-guarded.
|
||||
- **Files too**: previews, downloads and text reads in a remote case go over the same ssh connection (one `realpath` + `stat` probe, then a streamed `cat`, `Range` seeking included), so a clicked path opens the file on the machine the agent is on. Nothing is copied to the Codeman host; editing and Office previews answer a clear 400 instead of a misleading 404.
|
||||
|
||||
Set it up under **New Case → Remote** (host, user, identity file, optional jump host). Full design: [`docs/remote-sessions.md`](docs/remote-sessions.md).
|
||||
|
||||
@@ -647,8 +657,8 @@ These run for **every** request — before auth, even on the default no-password
|
||||
|
||||
### Input, files & headers
|
||||
|
||||
- **Schema-validated inputs** — every API body is checked with Zod v4 schemas; a `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `ANTIGRAVITY_*` / `GEMINI_*` / `GOOGLE_*` / `PI_*` env-prefix allowlist gates which settings each CLI can receive
|
||||
- **Path containment** — file routes `realpath` before boundary checks (no TOCTOU); `..`, absolute paths, and symlinks resolving outside the working dir are rejected. Caps: 10 MB text preview / 50 MB raw & download; `/api/download` blocklists sensitive paths (`.env`, `*credentials*`, `~/.ssh/`, `.aws/credentials`). SVG/HTML is served `octet-stream` + `nosniff` + attachment so it downloads rather than executes
|
||||
- **Schema-validated inputs** — every API body is checked with Zod v4 schemas; a `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `ANTIGRAVITY_*` / `GEMINI_*` / `GOOGLE_*` / `PI_*` / `GROK_*` / `XAI_*` / `DSH_*` / `DEEPSEEK_*` / `OMP_*` env-prefix allowlist gates which settings each CLI can receive, and the keys that could redirect a CLI's traffic (base URLs, config homes) are clamped for non-admin users
|
||||
- **Path containment** — file routes `realpath` before boundary checks (no TOCTOU); `..`, absolute paths, and symlinks resolving outside the working dir are rejected. Caps: 10 MB text preview / 2 GB raw & download (`CODEMAN_MAX_DOWNLOAD_BYTES`; bodies stream and answer `Range` requests, so the cap is a sanity bound rather than memory protection); `/api/download` blocklists sensitive paths (`.env`, `*credentials*`, `~/.ssh/`, `.aws/credentials`). SVG/HTML is served `octet-stream` + `nosniff` + attachment so it downloads rather than executes
|
||||
- **Security headers** — `Content-Security-Policy` (`default-src 'self'`, every exception enumerated), `X-Content-Type-Options: nosniff`, `X-Frame-Options: SAMEORIGIN`, HSTS over HTTPS, and CORS reflected **only** for `localhost` / `127.0.0.1` / `::1`
|
||||
|
||||
### Supply chain & isolation
|
||||
@@ -698,6 +708,10 @@ The web UI remains the primary surface; see **[docs/tui.md](docs/tui.md)** for t
|
||||
| `Ctrl/Cmd +` / `-` | Font size |
|
||||
| `Ctrl/Cmd+?` | Keyboard help |
|
||||
| `Shift+Enter` | Insert newline (sent to terminal) |
|
||||
| `Shift+drag` | Select text in a pane whose mouse events go to the CLI |
|
||||
| Right-click | Copy the selection (the native menu stays when nothing is selected) |
|
||||
| `Shift+Wheel` | Scroll the local scrollback while the wheel is forwarded to the CLI |
|
||||
| `Ctrl+Z` | Swallowed in agent sessions so a running CLI cannot be suspended; normal job control in a shell |
|
||||
| `Escape` | Close panels & modals |
|
||||
|
||||
---
|
||||
@@ -762,7 +776,7 @@ Those `DONE_<task>_<random>` strings are the skill's **split marker** trick, and
|
||||
| --------------------------------------------------------------------- | --------------------------------------------------------------------------------------------- |
|
||||
| [`SKILL.md`](skills/codeman/SKILL.md) | Safety rules, the ready-made fast path (spawn N workers, task them, collect), and the verb index. Always loaded. |
|
||||
| [`reference/verbs.md`](skills/codeman/reference/verbs.md) | The 14 verbs in detail: readiness, send-and-wait, markers, interrupts, cleanup. On demand. |
|
||||
| [`reference/recipes.md`](skills/codeman/reference/recipes.md) | 6 worked multi-worker flows (fan-out, blocked-worker watch, messaging fan-out). On demand. |
|
||||
| [`reference/recipes.md`](skills/codeman/reference/recipes.md) | 8 worked flows: claude, DeepSeek Harness and shell workers, fan-out, blocked-worker watch, messaging fan-out. On demand. |
|
||||
| [`reference/endpoints.md`](skills/codeman/reference/endpoints.md) | Full endpoint tables, error codes, per-mode signal table, capacity limits. On demand. |
|
||||
| [`reference/messaging.md`](skills/codeman/reference/messaging.md) | Talking to claude workers directly via Claude Code cross-session messaging. On demand. |
|
||||
|
||||
@@ -798,8 +812,8 @@ When a CLI runs in a Codeman-managed session, these environment variables are se
|
||||
4. **Response envelope.** Most endpoints return `{ "success": true, "data": … }` (errors: `{ "success": false, "error", "errorCode" }`). A few legacy GETs return bare bodies — **handle both** (`body.data ?? body`).
|
||||
5. **`/api/v1/*`** is a stable alias of `/api/*`.
|
||||
6. **Wait instead of polling, and don't treat a timeout as an error.** The wait endpoints answer with HTTP `200` and `wait.timedOut: true` when nothing happened in time, so loop over short waits (60s is the default) rather than issuing one long call, because tunnels cut idle connections. `wait.timeoutMs` tells you the timeout the server actually applied after clamping (600s ceiling).
|
||||
7. **Only `claude` sessions emit `stop` and `blocked`.** Those two come from Claude Code hooks; `shell` and the external CLIs (opencode/codex/gemini/antigravity/pi) accept only `idle`, `working` and `exit`. Asking for `stop` explicitly on those is a `400`; omitting `until` is always safe. ⚠️ On a `shell` session `idle` fires **once**, at startup, and never again, so send-and-wait there can only time out; synchronize hook-less sessions with a `wait-output` marker.
|
||||
7. **Only `claude` sessions emit `stop` and `blocked`.** Those two come from Claude Code hooks; `shell` and the external CLIs (opencode/codex/gemini/antigravity/omp) accept only `idle`, `working` and `exit`. Asking for `stop` explicitly on those is a `400`; omitting `until` is always safe. ⚠️ On a `shell` session `idle` fires **once**, at startup, and never again, so send-and-wait there can only time out; synchronize hook-less sessions with a `wait-output` marker.8. **Nothing reports "ready", so wait for it explicitly.** A new session answers `{"signal":"exit","immediate":true}` (that means *not started*, not *crashed*) until its PID exists, and a `claude` worker in a fresh case then sits on the CLI's trust dialog. Prompt it there and the wait resolves on `idle` in ~2s looking exactly like a finished turn, while the text sits stuck in the dialog. Recipe 2b below is the sequence that avoids it.
|
||||
7. **Only `claude` and `deepseek` sessions emit `stop` and `blocked`.** Those two come from hooks (Claude Code's own, and the DeepSeek Harness status bridge); `shell` and the other external CLIs (opencode/codex/gemini/antigravity/pi/grok/omp) accept only `idle`, `working` and `exit`. Asking for `stop` explicitly on those is a `400`; omitting `until` is always safe. ⚠️ On a `shell` session `idle` fires **once**, at startup, and never again, so send-and-wait there can only time out; synchronize hook-less sessions with a `wait-output` marker.
|
||||
8. **Nothing reports "ready", so wait for it explicitly.** A new session answers `{"signal":"exit","immediate":true}` (that means *not started*, not *crashed*) until its PID exists, and a `claude` worker in a fresh case then sits on the CLI's trust dialog. Prompt it there and the wait resolves on `idle` in ~2s looking exactly like a finished turn, while the text sits stuck in the dialog. Recipe 2b below is the sequence that avoids it.
|
||||
|
||||
### Recipes
|
||||
|
||||
@@ -866,9 +880,20 @@ curl -sG "$API/api/sessions/$SID/wait-output" \
|
||||
--data-urlencode "match=DONE_$N" --data-urlencode 'from=buffer' \
|
||||
--data-urlencode 'timeout=60000' | jq '.data.wait'
|
||||
|
||||
# 5. Read the terminal back. ⚠️ Use terminal?tail=, NOT /output: the latter's
|
||||
# textOutput is empty for every tmux-backed (i.e. every interactive) session.
|
||||
# tail counts BYTES, and what comes back is terminal data, ANSI included.
|
||||
# 5. Read the answer. claude / codex / deepseek sessions have last-response: it comes
|
||||
# from the transcript, not the screen, so no TUI frames or repaint noise.
|
||||
# ⚠️ Poll rather than read once: the transcript lands slightly after the stop
|
||||
# signal, so a read right after send-and-wait returns often comes back empty.
|
||||
for _ in $(seq 1 10); do
|
||||
TXT=$(curl -s "$API/api/sessions/$SID/last-response" | jq -r '.data.text')
|
||||
[ -n "$TXT" ] && break; sleep 1
|
||||
done
|
||||
printf '%s\n' "$TXT"
|
||||
|
||||
# 5b. Other modes (shell/opencode/gemini/antigravity/pi/grok/omp) have no transcript:
|
||||
# read the terminal. ⚠️ Use terminal?tail=, NOT /output: the latter's textOutput
|
||||
# is empty for every tmux-backed (i.e. every interactive) session. tail counts
|
||||
# BYTES, and what comes back is terminal data, ANSI included.
|
||||
curl -s "$API/api/sessions/$SID/terminal?tail=8000" | jq -r '.data.terminalBuffer'
|
||||
|
||||
# 6. Stream live events (session output, agent activity, status)
|
||||
@@ -914,7 +939,7 @@ Codeman registers Claude Code hooks that `POST /api/hook-event` (`permission_pro
|
||||
|
||||
## API
|
||||
|
||||
REST over Fastify — **~200 handlers across 21 route modules**, plus an SSE stream and a WebSocket terminal channel. All responses use the `ApiResponse<T>` envelope (`{success, data}` / `{success, error, errorCode}`); `/api/v1/*` is a stable alias. A representative subset:
|
||||
REST over Fastify — **~230 handlers across 25 route modules**, plus an SSE stream and a WebSocket terminal channel. All responses use the `ApiResponse<T>` envelope (`{success, data}` / `{success, error, errorCode}`); `/api/v1/*` is a stable alias. A representative subset:
|
||||
|
||||
### Sessions
|
||||
|
||||
@@ -925,11 +950,13 @@ REST over Fastify — **~200 handlers across 21 route modules**, plus an SSE str
|
||||
| `POST` | `/api/sessions/:id/input` | Send input (`{input, useMux?, clientId?, seq?, wait?, waitTimeout?}`: `clientId`+`seq` = exactly-once; `wait` blocks until the turn ends) |
|
||||
| `GET` | `/api/sessions/:id/terminal` | Read terminal output (`?tail=<bytes>`, `?full=1`); the read path for interactive sessions |
|
||||
| `GET` | `/api/sessions/:id/output` | Parsed one-shot output (`textOutput` is empty for tmux-backed sessions) |
|
||||
| `GET` | `/api/sessions/:id/last-response` | The last answer as clean text, read from the transcript (claude, codex, deepseek) |
|
||||
| `GET` | `/api/sessions/:id/wait` | Block until a signal fires (`?until=stop,idle,exit&timeout=&fresh=`); a timeout is a `200` |
|
||||
| `GET` | `/api/sessions/:id/wait-output` | Block until a literal string appears (`?match=&nocase=&from=now\|buffer&timeout=`) |
|
||||
| `GET` | `/api/sessions/unified` | Unified live + history list (Session Manager) — `?q=&limit=` |
|
||||
| `POST` | `/api/sessions/:id/pin` | Pin/unpin in the Session Manager (`{pinned}`) |
|
||||
| `PUT` | `/api/session-order` | Sync tab order across devices (`{order: [ids]}`) |
|
||||
| `POST` | `/api/sessions/:id/custom-model` | Restart the session's CLI on a saved custom endpoint (`{endpointId, modelId}`; `{clear: true}` returns to the native backend) |
|
||||
| `DELETE` | `/api/sessions/:id` | Delete session |
|
||||
|
||||
### Respawn
|
||||
@@ -978,6 +1005,7 @@ REST over Fastify — **~200 handlers across 21 route modules**, plus an SSE str
|
||||
| `GET` | `/api/system/update/check` | Check for a new release |
|
||||
| `POST` | `/api/system/update` | Self-update (git-clone installs) |
|
||||
| `POST` | `/api/clipboard` | Push text to all connected browsers (`{text}`) |
|
||||
| `GET` / `POST` | `/api/model-endpoints` | List / save custom OpenAI-compatible endpoints (`PUT` / `DELETE` `/:id`; admin-only in multi-user mode) |
|
||||
| `GET` | `/api/sessions/:id/run-summary` | Timeline + stats |
|
||||
|
||||
> **Building something on top of Codeman?** [`docs/extending-codeman.md`](docs/extending-codeman.md) is the integration guide: render your own UI as a tab, subscribe to the SSE event stream to react when an agent needs you, drive Codeman from a script, and the traps worth knowing before you start. Codeman has no plugin runtime on purpose, so an integration is just your own process talking HTTP.
|
||||
@@ -1014,8 +1042,8 @@ flowchart TB
|
||||
end
|
||||
|
||||
subgraph External["External"]
|
||||
CLI["AI CLI<br/><small>Claude Code / OpenCode / Codex / Antigravity / Gemini / Pi</small>"]
|
||||
CLI["AI CLI<br/><small>Claude Code / OpenCode / Codex / Antigravity / Gemini / OMP</small>"] BG["Background Agents<br/><small>(Task tool)</small>"]
|
||||
CLI["AI CLI<br/><small>Claude Code / OpenCode / Codex / Antigravity / Gemini / Pi / Grok / DeepSeek / OMP</small>"]
|
||||
BG["Background Agents<br/><small>(Task tool)</small>"]
|
||||
end
|
||||
end
|
||||
|
||||
@@ -1081,7 +1109,7 @@ Full details: [`docs/archive/code-structure-findings.md`](docs/archive/code-stru
|
||||
|
||||
[](https://www.npmjs.com/package/xterm-zerolag-input)
|
||||
|
||||
Instant keystroke feedback overlay for xterm.js. Eliminates perceived input latency over high-RTT connections by rendering typed characters immediately as a pixel-perfect DOM overlay. Zero dependencies, 6.1 kB gzipped, configurable prompt detection, CJK/emoji wide-character support, full state machine with 175 tests.
|
||||
Instant keystroke feedback overlay for xterm.js. Eliminates perceived input latency over high-RTT connections by rendering typed characters immediately as a pixel-perfect DOM overlay. Zero dependencies, 6.1 kB gzipped, configurable prompt detection, CJK/emoji wide-character support, full state machine with 238 tests.
|
||||
|
||||
```bash
|
||||
npm install xterm-zerolag-input
|
||||
|
||||
+207
-51
@@ -5,7 +5,7 @@
|
||||
<h2 align="center">AI 编程智能体的任务控制中心</h2>
|
||||
|
||||
<p align="center">
|
||||
<em>Claude Code • OpenCode • Codex • Antigravity • Gemini • Pi • Grok • 终端 —— 统一仪表盘 • 任意设备</em>
|
||||
<em>Claude Code • OpenCode • Codex • Antigravity • Gemini • Pi • Grok • DeepSeek • OMP • 终端 —— 统一仪表盘 • 任意设备</em>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
@@ -17,6 +17,8 @@
|
||||
<a href="https://nodejs.org/"><img src="https://img.shields.io/badge/Node.js-22%2B-22c55e?style=flat-square&logo=node.js&logoColor=white" alt="Node.js 22+"></a>
|
||||
<a href="https://www.typescriptlang.org/"><img src="https://img.shields.io/badge/TypeScript-5.9-3b82f6?style=flat-square&logo=typescript&logoColor=white" alt="TypeScript 5.9"></a>
|
||||
<a href="https://fastify.dev/"><img src="https://img.shields.io/badge/Fastify-5.x-1e3a5f?style=flat-square&logo=fastify&logoColor=white" alt="Fastify"></a>
|
||||
<a href="https://www.npmjs.com/package/aicodeman"><img src="https://img.shields.io/npm/v/aicodeman?style=flat-square&label=npm&color=22c55e" alt="npm version"></a>
|
||||
<a href="https://github.com/Ark0N/Codeman/stargazers"><img src="https://img.shields.io/github/stars/Ark0N/Codeman?style=flat-square&color=eab308" alt="GitHub stars"></a>
|
||||
<a href="https://github.com/Ark0N/Codeman/graphs/contributors"><img src="https://img.shields.io/github/contributors/Ark0N/Codeman?style=flat-square&color=3b82f6" alt="Contributors"></a>
|
||||
<a href="https://github.com/Ark0N/Codeman/commits/master"><img src="https://img.shields.io/github/commit-activity/t/Ark0N/Codeman?style=flat-square&color=1e3a5f" alt="Total commits"></a>
|
||||
</p>
|
||||
@@ -25,12 +27,10 @@
|
||||
<img src="docs/images/subagent-demo-20260724.gif" alt="Codeman — 并行子智能体可视化" width="900">
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/codeman-tour-20260724.png" alt="Codeman 仪表盘导览:按项目分组的会话标签页、一键 Run 启动新智能体、页头实时用量" width="900">
|
||||
</p>
|
||||
|
||||
> 本文档由英文版 [`README.md`](README.md) 翻译而来。如有出入,以英文版为准。
|
||||
|
||||
**Codeman** 是一个自托管的 AI 编程智能体任务控制中心。它在持久化的 tmux 会话里拉起 Claude Code、OpenCode、Codex、Antigravity、Gemini、Pi、Grok、DeepSeek Harness 或 OMP,把真实的终端流式传到任意浏览器,并在你离开之后让智能体继续干活:空闲时重新提示、用量限额重置后自动续跑、按计划执行任务,还能实时展示每一个后台智能体的工作。
|
||||
|
||||
一行命令即可安装(macOS 和 Linux,Windows 通过 WSL):
|
||||
|
||||
```bash
|
||||
@@ -44,6 +44,17 @@ codeman web
|
||||
|
||||
安装器在每次系统改动前都会先询问;重跑同一条命令即可原地更新。详见[快速开始 — 安装](#快速开始--安装)。
|
||||
|
||||
- **一个仪表盘,九个 CLI**:每个会话可选 [Claude Code、OpenCode、Codex、Antigravity、Gemini、Pi、Grok、DeepSeek 或 OMP](#更多特性)(外加普通 shell),在本机、[Docker 容器](#隔离的-docker-会话)或 [SSH 远程主机](#远程-ssh-会话)上运行,你自己的仪表盘也能作为 [Web 标签页](#更多特性)并排打开
|
||||
- **真正的手机友好**:[触控优化的终端](#移动端优化的-web-ui),即时本地回显、二维码登录、滑动导航与推送通知
|
||||
- **睡觉时也在跑**:[空闲检测 + 重生循环](#重生控制器respawn-controller),订阅限额重置后自动续跑,支持 24 小时以上的无人值守运行
|
||||
- **看见智能体在想什么**:每个子智能体和团队成员都有[实时浮动窗口](#实时智能体可视化),附带实时活动记录
|
||||
- **什么都不会丢**:tmux 让会话挺过重启和断网,输入精确一次送达,完整的回滚缓冲区回放
|
||||
- **自托管、私有**:默认仅环回、MIT 许可、无遥测,完全运行在你自己的机器上
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/codeman-tour-20260724.png" alt="Codeman 仪表盘导览:按项目分组的会话标签页、一键 Run 启动新智能体、页头实时用量" width="900">
|
||||
</p>
|
||||
|
||||
---
|
||||
|
||||
## 快速开始 — 安装
|
||||
@@ -52,13 +63,14 @@ codeman web
|
||||
curl -fsSL https://getcodeman.com/install | bash
|
||||
```
|
||||
|
||||
该脚本会在缺失时自动安装 Node.js 和 tmux,把 Codeman 克隆到 `~/.codeman/app` 并完成构建。几点须知:
|
||||
该脚本会在缺失时自动安装 Node.js、tmux 和一套构建工具链(node-pty 没有 Linux 预编译包,需要从源码编译),把 Codeman 克隆到 `~/.codeman/app` 并完成构建。几点须知:
|
||||
|
||||
- **先询问,后改动。** 所有系统级改动(安装软件包、下载 AI CLI)都会先征求确认;结束时的菜单可选择:直接在本终端运行、安装为后台服务(systemd/launchd,开机自启),或暂不启动。不选就不会有任何后台进程。
|
||||
- **怎么访问,由你决定。** 安装器提供三种到达仪表盘的方式:**Tailscale**(环回绑定,由 `tailscale serve` 代理,得到带真实证书的 `https://<机器名>.<tailnet>.ts.net`,用你的 tailnet 当登录,无需密码)、**局域网内任意设备**(`0.0.0.0`,会提示设置一个强烈推荐的密码),或**仅本机**(`127.0.0.1`,最安全)。绑定网络却跳过密码需要显式确认,并以醒目警告收尾。高亮的默认项反映机器上已有的状态(已在用 Tailscale 时默认 Tailscale,重跑时沿用现有绑定),直接回车绝不会引入新软件。手动运行的 `codeman web` 仍默认仅环回。
|
||||
- **重跑即更新。** 再次运行同一条命令即可原地更新已完成的安装:`~/.codeman/app` 中的本地改动会被 stash(绝不丢弃),运行中的服务会自动重启并校验。若首次安装中途失败,重跑会继续完成完整的安装流程。也可以使用 `install.sh update` 与 `install.sh uninstall`。
|
||||
- **CI / 无终端环境:** 没有终端时,涉及系统改动的步骤会带着说明中止,而不是静默执行;在自动化场景设置 `CODEMAN_NONINTERACTIVE=1` 即可批准这些步骤。
|
||||
|
||||
你至少需要安装一个 AI 编程 CLI —— [Claude Code](https://docs.anthropic.com/en/docs/claude-code)、[OpenCode](https://opencode.ai)、[Codex](https://developers.openai.com/codex/cli)、[Antigravity](https://antigravity.google)、[Gemini CLI](https://github.com/google-gemini/gemini-cli)、[Pi](https://pi.dev)、[Grok Build](https://github.com/xai-org/grok-build)、[DeepSeek Harness](https://github.com/deepseek-ai/deepseek-harness) 或 [OMP](https://github.com/can1357/oh-my-pi)(任意组合均可;自 Google 面向消费者停售后,Gemini CLI 仅限企业版,Antigravity 是其继任者)。安装器会自动检测这九个中已安装的任意一个;若一个都没有,会提供安装 Claude Code 或 OpenCode 的选项,也可以选择跳过、稍后自行安装。安装完成后:
|
||||
你至少需要安装一个 AI 编程 CLI —— [Claude Code](https://docs.anthropic.com/en/docs/claude-code)、[OpenCode](https://opencode.ai)、[Codex](https://developers.openai.com/codex/cli)、[Antigravity](https://antigravity.google)、[Gemini CLI](https://github.com/google-gemini/gemini-cli)、[Pi](https://pi.dev)、[Grok Build](https://github.com/xai-org/grok-build)、[DeepSeek Harness](https://github.com/deepseek-ai/deepseek-harness) 或 [OMP](https://github.com/can1357/oh-my-pi)(任意组合均可;自 Google 面向消费者停售后,Gemini CLI 仅限企业版,Antigravity 是其继任者)。安装器会自动检测这九个中已安装的任意一个;若一个都没有,会给出一个菜单让你安装其中任意一个(DeepSeek 除外,它的 npm 包只装一个启动器,没有可运行的 profile),也可以选择跳过、稍后自行安装。安装完成后:
|
||||
|
||||
```bash
|
||||
codeman web
|
||||
@@ -72,12 +84,34 @@ codeman users add alice --admin # 创建第一个管理员账号
|
||||
codeman web --multiuser # 命名登录 + 按用户隔离的案例空间
|
||||
```
|
||||
|
||||
**更喜欢 Docker Compose?** `docker/` 里附带一套本地镜像的 Compose 部署:把 `docker/.env.example` 复制为 `docker/.env`,设置 `CODEMAN_PASSWORD`,然后在 Linux 上运行 `bash docker/Start-Codeman.sh`。Codeman 自己跑在容器里,并通过宿主机的 socket 把 Docker 案例作为并列容器拉起。更新之后请再跑一次这个脚本,而不是直接 `docker compose up`,这样重建的镜像、刷新的卷和新的入口脚本会一起就位。直接的 Compose 命令、存储与网络选项见 [Docker 部署指南](docker/README.md)(英文)。
|
||||
|
||||
详见下文[多用户模式](#多用户模式可选启用)。
|
||||
|
||||
<details>
|
||||
<summary><strong>作为后台服务运行</strong></summary>
|
||||
<summary><strong>让它在后台一直运行</strong></summary>
|
||||
|
||||
安装器结尾的菜单(选项 2)可以帮你完成这一步,并在宣告成功前校验服务确实已启动。如需手动配置:
|
||||
想让它活过你启动它的那个 shell,而且什么都不用配置:
|
||||
|
||||
```bash
|
||||
codeman web -d # 脱离终端;日志写到 ~/.codeman/web.log
|
||||
codeman web --status # 是否在运行,pid 是多少
|
||||
codeman web --stop # 优雅的 SIGTERM;智能体继续留在 tmux 里运行
|
||||
```
|
||||
|
||||
`-d` 会等到服务器真正应答后才报告成功,并且拒绝在同一个数据目录上启动第二个(两个服务器共用一个 tmux socket 会互相附着对方的会话)。
|
||||
|
||||
想让它在重启后自动回来,就装成服务。安装器结尾的菜单(选项 2)会替你完成;`codeman service` 是 `npm i -g aicodeman` 安装的等价物:
|
||||
|
||||
```bash
|
||||
codeman service install # systemd 用户单元(Linux)或 LaunchAgent(macOS)
|
||||
codeman service status
|
||||
codeman service uninstall
|
||||
```
|
||||
|
||||
`service install` 会把你当前的 PATH 写进单元文件,这比听起来重要得多:launchd 只给任务 `/usr/bin:/bin:/usr/sbin:/sbin`,所以手写的 plist 根本找不到 Homebrew 或 nvm 装的 `node`、`tmux` 或 `claude`。它绝不会把 `CODEMAN_PASSWORD` 复制进单元文件;服务需要认证的话请自行添加。
|
||||
|
||||
如需手动编写单元文件:
|
||||
|
||||
**Linux(systemd):**
|
||||
|
||||
@@ -177,17 +211,17 @@ Codeman 依赖 tmux,因此 Windows 用户需要 [WSL](https://learn.microsoft.
|
||||
<tr><td>在手机上手打密码</td><td><b>扫二维码 —— 即时认证</b></td></tr>
|
||||
</table>
|
||||
|
||||
- **键盘配件栏** —— 在虚拟键盘上方提供 `/init`、`/clear`、`/compact` 快捷按钮;破坏性命令需双击确认,绝不误触
|
||||
- **键盘配件栏** —— 在虚拟键盘上方提供 `/init`、`/clear`、`/compact` 快捷按钮;破坏性命令需双击确认,绝不误触;在 Codex 会话上还会显示 `⇧←` / `⇧→`(Shift+Left / Shift+Right:编辑上一条排队的消息 / 在提示栈里回退)
|
||||
- **独立的 Enter 按钮** —— 以按键方式回放,先冲刷本地回显缓冲的文本,不会让内容滞留在屏幕上
|
||||
- **滑动导航与智能键盘处理** —— 左右滑动切换会话;键盘弹出时工具栏与终端整体上移(`visualViewport` API)
|
||||
- **为手机而生** —— 刘海与 Home 指示条的安全区适配、44px 触控目标、底部抽屉式 case 选择器、原生惯性滚动
|
||||
- **为手机而生** —— 刘海与 Home 指示条的安全区适配、44px 触控目标、底部抽屉式 case 选择器、原生惯性滚动;折叠屏手机(iPhone Duo)上对话框会避开铰链,开合设备也绝不会被误判成键盘弹出
|
||||
|
||||
```bash
|
||||
codeman web --https
|
||||
# 在手机上打开:https://<你的IP>:3000
|
||||
```
|
||||
|
||||
> `localhost` 走纯 HTTP 即可。从其他设备访问时请使用 `--https`,或使用 [Tailscale](https://tailscale.com/)(推荐)—— 它提供私有网络,让你无需 TLS 证书即可从手机访问 `http://<tailscale-ip>:3000`。
|
||||
> `localhost` 走纯 HTTP 即可。从其他设备访问时请使用 `--https`,或使用 [Tailscale](https://tailscale.com/)(推荐):安装器可以替你配好(在网络访问提示处选择 **Tailscale**,或在已有安装上运行 `bash ~/.codeman/app/install.sh tailscale`)。这样你会得到带真实证书的 `https://<你的机器>.<tailnet>.ts.net`:只对你的 tailnet 可见、无需密码,手机上的 PWA 安装和推送通知也都能用。
|
||||
|
||||
### 安全的二维码认证
|
||||
|
||||
@@ -210,6 +244,8 @@ codeman web # localhost:3000(仅环回 —— 安全默
|
||||
codeman web --port 8080 # 自定义端口(或设置 CODEMAN_PORT)
|
||||
codeman web --https # 自签名 TLS(仅远程访问时需要)
|
||||
codeman web -H 0.0.0.0 # 绑定局域网 —— 必须设置 CODEMAN_PASSWORD(见「安全」)
|
||||
codeman web -d # 脱离终端:关掉 shell 也在跑(--status、--stop)
|
||||
codeman service install # systemd/launchd 服务:重启后自动回来
|
||||
```
|
||||
|
||||
打开打印出的 URL。整个页面是一个单一仪表盘;下面的一切都在这里完成。
|
||||
@@ -220,16 +256,16 @@ codeman web -H 0.0.0.0 # 绑定局域网 —— 必须设置 CODEMAN_
|
||||
|
||||
| 字段 | 作用 |
|
||||
| ---------------------- | ------------------------------------------------------------------------------------------- |
|
||||
| **工作目录 / case** | 智能体操作的文件夹。「case」就是一个 Codeman 记住的命名工作目录。 |
|
||||
| **CLI / 运行模式** | `Claude`(默认)、`OpenCode`、`Codex`、`Antigravity`、`Gemini`、`Pi`、`Grok` 或 `Terminal`(普通 shell)。 |
|
||||
| **模型** | 每会话模型(App Settings → Claude Model)。软默认值 —— 会话内 `/model` 依然有效。 |
|
||||
| **工作目录 / case** | 智能体操作的文件夹。「case」就是一个 Codeman 记住的命名工作目录。**Add Case** 可以从零创建、链接一个已有文件夹,或把一个 GitHub 仓库直接克隆成 case(**Clone Repo**)。 |
|
||||
| **CLI / 运行模式** | `Claude`(默认)、`OpenCode`、`Codex`、`Antigravity`、`Gemini`、`Pi`、`Grok`、`DeepSeek`、`OMP` 或 `Terminal`(普通 shell)。 |
|
||||
| **模型** | 每会话模型(App Settings → Models → New Claude sessions)。软默认值 —— 会话内 `/model` 依然有效。 |
|
||||
| **Effort / Ultracode** | 推理力度(`low`–`max`),或用 `ultracode` 开启动态多智能体工作流。随时可用 `/effort` 切换。 |
|
||||
|
||||
点击启动 —— Codeman 通过真实 PTY 拉起 CLI,并经 SSE 流式传输到你的浏览器。
|
||||
|
||||
### 3. 读懂仪表盘
|
||||
|
||||
- **标签(顶部)** —— 每个会话一个。`Alt+1`–`9` 跳转,`Ctrl+Tab` 下一个,拖拽排序(标签顺序会跨设备同步)。
|
||||
- **标签(顶部)** —— 每个会话一个。`Alt+1`–`9` 跳转,`Ctrl+Tab` 下一个,拖拽排序(标签顺序会跨设备同步)。更喜欢列表?**App Settings → Appearance → Tabs** 可以把它挪进左侧边栏(带筛选框,`Alt+B` 折叠)或一条竖向导轨,导轨的行按活动状态排序:先是等你处理的,然后是跑得最久的,最后是刚刚安静下来的。
|
||||
- **终端(中央)** —— 真实的 `xterm.js` 终端;完整 TUI 正常渲染。直接输入并按 **Enter** 发送。`Shift+Enter` 插入换行。
|
||||
- **侧边面板** —— Respawn、Orchestrator、Cron、Subagents、Settings(从工具栏切换)。
|
||||
|
||||
@@ -237,8 +273,10 @@ codeman web -H 0.0.0.0 # 绑定局域网 —— 必须设置 CODEMAN_
|
||||
|
||||
- **直接在终端输入提示** —— 即使跨越重连,输入也是精确一次送达(连接中断绝不会丢失或重复发送提示)。
|
||||
- **粘贴或拖放图片**,直接进入会话。
|
||||
- **语音输入** —— `Ctrl+Shift+V`(Deepgram Nova-3,自动静音停止)。
|
||||
- **附件** —— 注册外部文件/文档,并内联预览 Office/PDF。
|
||||
- **语音输入** —— `Ctrl+Shift+V`(Deepgram Nova-3,或者直接用这台机器的 Claude Code 登录、不需要任何 API key;自动静音停止)。
|
||||
- **附件** —— 注册外部文件/文档,并内联预览 Office/PDF;智能体打印出的任何文件路径都可以点击,终端里和对话视图里都行。
|
||||
- **需要你的时候** —— 标签会变黄(等待输入)或变红(有个问题挡住了它)。**审批收件箱(Approvals Inbox)**(可选启用)把所有会话里等着你的提示排成一个队列,可以从页头的铃铛或手机首页直接作答;🧠 **Read My Mind**(可选启用)会根据这个 case 的目标和最近的工作替你起草下一条提示。
|
||||
- **看到什么就能复制什么** —— `Shift+拖动` 在 CLI 接管了鼠标时也能选中文本,右键复制选中内容,自动复制(Auto Copy,可选启用)在松开鼠标的瞬间就复制。
|
||||
|
||||
### 5. 让它自主运行
|
||||
|
||||
@@ -246,7 +284,7 @@ codeman web -H 0.0.0.0 # 绑定局域网 —— 必须设置 CODEMAN_
|
||||
| ---------------- | --------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------ |
|
||||
| **Respawn** | 长时间无人值守运行 —— 空闲/限额时自动重启 CLI,带自适应时序。预设:`solo-work`、`overnight-autonomous` 等 | Respawn 标签页 |
|
||||
| **Orchestrator** | 把一个目标变成分阶段计划,并跨多个智能体推动完成。 | 编排器面板 |
|
||||
| **Cron** | 已保存的、命名的定时任务(`once`/`interval`/`daily`/`weekly`),到期时拉起会话并发送提示。 | ⏰ Cron 按钮(可选启用:App Settings → Display → Header Displays) |
|
||||
| **Cron** | 已保存的、命名的定时任务(`once`/`interval`/`daily`/`weekly`),到期时拉起会话并发送提示。 | ⏰ Cron 按钮(可选启用:App Settings → Header & Panels → Scheduling) |
|
||||
| **Auto-resume** | 订阅限额重置后自动继续。 | Respawn 标签页(顶部) |
|
||||
|
||||
### 6. 随时随地访问
|
||||
@@ -257,8 +295,9 @@ codeman web -H 0.0.0.0 # 绑定局域网 —— 必须设置 CODEMAN_
|
||||
|
||||
### 7. 运维与维护
|
||||
|
||||
- **App Settings** —— 模型、effort、权限启动模式、主题/皮肤、通知、显示开关、各 CLI 的专属选项,以及跨设备同步的自定义显示名称和按设备保存的英文/简体中文界面语言。
|
||||
- **自更新** —— git-clone 安装可在 **Settings → Updates** 中原地更新。
|
||||
- **App Settings** —— 模型、effort、权限启动模式、主题/皮肤、终端字体与字重、入场动画、通知、显示开关、各 CLI 的专属选项,以及跨设备同步的自定义显示名称和按设备保存的英文/简体中文界面语言。
|
||||
- **让它在后台运行** —— `codeman web -d` 脱离你的 shell(`--status`、`--stop`);`codeman service install` 把它装成 systemd 用户单元 / macOS LaunchAgent,重启后自动回来。两者都会先确认服务器真正应答再报告成功,也都拒绝在同一个数据目录上启动第二个服务器。见[让它在后台一直运行](#快速开始--安装)。
|
||||
- **自更新** —— git-clone 安装可在 **App Settings → System → Updates** 中原地更新。
|
||||
- **部署你自己的改动** —— 见[开发](#开发)。
|
||||
|
||||
> ⚠️ **安全提示:** 如果你正在 Codeman 受管会话*内部*工作(`echo $CODEMAN_MUX` → `1`),绝不要直接运行 `tmux kill-session` / `pkill claude` —— 请使用 Web UI 或 `./scripts/tmux-manager.sh`。
|
||||
@@ -373,6 +412,14 @@ codeman web --title-hostname dev-box # codeman:dev-box(用于覆盖嘈
|
||||
| **110k tokens** | 自动 `/compact` | 上下文被摘要,工作继续 |
|
||||
| **140k tokens** | 自动 `/clear` | 以 `/init` 全新开始 |
|
||||
|
||||
### 标签提醒(Tab Alerts)
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/tab-alerts-glow-20260815.gif" alt="会话标签:一个普通的活动标签,旁边是黄色的等待输入标签和红色的需要决定标签,都带着呼吸式光晕" width="900">
|
||||
</p>
|
||||
|
||||
每个标签一眼就能看出状态。运行中的会话保持绿色状态点。会话停下来等待输入时,标签变**黄**:稳定的描边、着色的背景、黄色的点,上面叠一层缓慢的呼吸光晕。当权限提示或提问**挡住**了智能体,标签变**红**,脉动更快。底色永远不会闪灭,所以哪怕只瞥一眼(或截一张图)也能读到真实状态;标签被选中时描边依然可见,页面刷新后会从服务端重新装载待处理的提醒,因此一个被挡住的会话绝不可能藏在一个看起来正常的标签后面。
|
||||
|
||||
### 通知
|
||||
|
||||
当会话需要关注时实时桌面提醒 —— `permission_prompt` 与 `elicitation_dialog` 触发关键的红色标签闪烁,`idle_prompt` 触发黄色闪烁。点击任意通知即可直接跳转到相关会话。Hook 按 case 目录自动配置。
|
||||
@@ -393,17 +440,24 @@ PTY 输出 → 16ms 服务端批处理 → DEC 2026 包裹 → SSE → 客户端
|
||||
|
||||
## 更多特性
|
||||
|
||||
- **自更新** —— systemd/launchd 管理下的 git-clone 安装可在 **App Settings → Updates** 中原地更新:它会检测最新发行版,自动暂存(stash)脏工作树,并在服务重启期间流式展示构建进度(npm 安装会被报告为不可更新)
|
||||
- **多 CLI** —— 每个会话可选 **Claude Code**、**OpenCode**、**Codex**、**Antigravity**、**Gemini**、**Pi** 或 **Grok**;环境变量前缀自动隔离(`CLAUDE_CODE_*`、`OPENCODE_*`、`CODEX_*`、`ANTIGRAVITY_*`、`PI_*`、`GROK_*`/`XAI_*` 与 `GEMINI_*`/`GOOGLE_*`)。详见 [`docs/opencode-integration.md`](docs/opencode-integration.md)、[`docs/pi-integration.md`](docs/pi-integration.md) 与 [`docs/grok-integration.md`](docs/grok-integration.md)
|
||||
- **Docker 会话** —— 在隔离且加固的容器中运行案例。**Create New** 上勾选一个复选框即可用合理的默认值启动容器并在其中启动智能体;同一案例的多个会话共享一个容器;可将容器连同工作区导出为可移植的 `.tar.gz`,迁移到另一台机器。详见 [`docs/docker-cases.md`](docs/docker-cases.md)
|
||||
- **远程 SSH 会话**:把案例指向另一台机器,让智能体在那里一个持久的远程 tmux 中运行:SSH 断连不中断任务、自动重连,还能发现并附着主机上已在运行的会话。详见 [`docs/remote-sessions.md`](docs/remote-sessions.md)
|
||||
- **后台守护进程与服务安装** —— `codeman web -d` 以脱离终端的方式运行服务器,带 pid 文件、`~/.codeman/web.log` 和经过校验的启动(它会轮询到服务器应答为止,所以端口冲突绝不会被当成成功);`codeman service install` 写入一个 systemd 用户单元(Linux)或 LaunchAgent(macOS),并把你 shell 的 PATH 一并写进去,这样 nvm 或 Homebrew 装的 `node`、`tmux` 和 `claude` 才真的找得到。机密永远不会写进单元文件
|
||||
- **自更新** —— systemd/launchd 管理下的 git-clone 安装可在 **App Settings → System → Updates** 中原地更新:它会检测最新发行版,自动暂存(stash)脏工作树,并在服务重启期间流式展示构建进度(npm 安装会被报告为不可更新)
|
||||
- **把 GitHub 仓库克隆成 case** —— 在 **Add Case → Clone Repo** 里粘贴一个仓库 URL,Codeman 会把它克隆到 `~/codeman-cases/<name>` 并注册为普通 case,随时可以跑智能体。输入时它会预检 URL(告诉你能否匿名克隆,并为可选的分支/标签字段提供仓库真实的分支与标签),从 URL 里填好 case 名,还让你选 Run 按钮该用哪个 CLI。支持 `https://` 的公开仓库;Codeman 绝不收集或保存凭据
|
||||
- **多 CLI** —— 每个会话可选 **Claude Code**、**OpenCode**、**Codex**、**Antigravity**、**Gemini**、**Pi**、**Grok**、**DeepSeek Harness** 或 **OMP**;环境变量前缀自动隔离(`CLAUDE_CODE_*`、`OPENCODE_*`、`CODEX_*`、`ANTIGRAVITY_*`、`GEMINI_*`/`GOOGLE_*`、`PI_*`、`GROK_*`/`XAI_*`、`DSH_*`/`DEEPSEEK_*` 与 `OMP_*`)。详见 [`docs/opencode-integration.md`](docs/opencode-integration.md)、[`docs/pi-integration.md`](docs/pi-integration.md)、[`docs/grok-integration.md`](docs/grok-integration.md)、[`docs/deepseek-integration.md`](docs/deepseek-integration.md) 与 [`docs/omp-integration.md`](docs/omp-integration.md)
|
||||
- **自定义模型端点**(1.29.0 新增,目前仅 HTTP API)—— 让某个会话的 CLI 指向任意 OpenAI 兼容端点,而不是它自己的官方后端:本地的 llama.cpp、llama-swap、Ollama 或 vLLM 机器,也可以是 Azure AI Foundry、OpenRouter 这类云端网关。端点只需保存一次(`POST /api/model-endpoints`,模型列表从它的 `/v1/models` 自动发现),再应用到会话(`POST /api/sessions/:id/custom-model`),CLI 就会在原地重启并接上该端点。Claude、OpenCode、Pi、Grok 与 OMP 已实测通过;Codex、Gemini 与 DeepSeek 存在已记录的缺口,Antigravity 没有可用机制。工具栏选择器是下一步。详见 [`docs/custom-model-endpoints.md`](docs/custom-model-endpoints.md)
|
||||
- **Web 标签页** —— 把 Grafana、Uptime Kuma、一个 Vite 开发服务器或任何仪表盘 URL 作为标签页打开在会话旁边(Run 下拉菜单 → **Web / URL** → **Add URL**)。仪表盘通过 Codeman 自己的源代理,因此 `http://` 目标在手机上走 HTTPS 也能用、走隧道也能用;单页应用能在自己的路径上正常路由,页面自己重载后也能自行恢复。智能体打印出的 `localhost` 链接会自动以 Web 标签页打开。详见 [`docs/web-tabs.md`](docs/web-tabs.md)
|
||||
- **Docker 会话** —— 在隔离且加固的容器中运行 case。**Create New** 上勾选一个复选框即可用合理的默认值启动容器并在其中启动智能体;同一 case 的多个会话共享一个容器,也可以把 case 挂到你已经在跑的容器上;可将容器连同工作区导出为可移植的 `.tar.gz`,迁移到另一台机器。详见 [`docs/docker-cases.md`](docs/docker-cases.md)
|
||||
- **远程 SSH 会话** —— 把 case 指向另一台机器,让智能体在那里一个持久的远程 tmux 中运行:SSH 断连不中断任务、自动重连,还能发现并附着主机上已在运行的会话;文件预览与下载走同一条 ssh 连接。详见 [`docs/remote-sessions.md`](docs/remote-sessions.md)
|
||||
- **Effort 与 Ultracode** —— 设置每会话的默认 effort(`low`–`max`),或启用 **ultracode**(动态多智能体工作流)。这些都只是软默认值 —— 会话中可随时用 `/effort` 切换。扩展思考预算也可配置
|
||||
- **语音输入** —— 用 Deepgram Nova-3 口述提示(带 Web Speech API 回退):切换录音、自动静音停止、实时音量表(`Ctrl+Shift+V`)
|
||||
- **语音输入** —— 用 Deepgram Nova-3 口述提示,或者干脆用这台机器的 Claude Code 登录、不需要任何 API key(App Settings → Voice;带 Web Speech API 回退):切换录音、自动静音停止、实时音量表(`Ctrl+Shift+V`)
|
||||
- **图像输入** —— 直接把图片粘贴或拖放进会话
|
||||
- **手势控制** _(可选)_ —— 一个 MediaPipe 手部追踪叠加层,可徒手抓取/拖动会话窗口并捏合按钮。用 `CODEMAN_GESTURE=1` + App Settings → Display 启用
|
||||
- **手势控制** _(可选)_ —— 一个 MediaPipe 手部追踪叠加层,可徒手抓取/拖动会话窗口并捏合按钮。用 `CODEMAN_GESTURE=1` + App Settings → Terminal & Input 启用
|
||||
- **多显示器横跨** _(macOS)_ —— 一键打开一个横跨所有显示器最大化的浏览器窗口,让浮动的智能体/手势面板可以跨越物理拼接缝
|
||||
- **文件查看器按钮** _(可选)_ —— 头部新增一个按钮,一键切换内置文件浏览器面板;在 App Settings → Display → Header Displays 中启用
|
||||
- **CJK / 输入法支持** —— 完整支持中文 / 日文 / 韩文的组合输入
|
||||
- **文件查看器按钮** _(可选)_ —— 页头新增一个按钮,一键切换内置文件浏览器面板;在 App Settings → Header & Panels → Header buttons 中启用
|
||||
- **CJK / 输入法支持** —— 完整支持中文 / 日文 / 韩文的组合输入,Ctrl、Alt 修饰的导航键也会原样透传给 CLI
|
||||
- **页头里的套餐用量** —— 页头实时显示 Claude 订阅用量(5 小时窗口与每周窗口),数据来自 Codeman 在拉起 `claude` 时临时交给它的 statusline 导出器,绝不会写进你的设置文件;Codex 的限额则来自它自己的 app-server。按设备生效:桌面默认开,手机默认关
|
||||
- **会话列表,随你摆** —— 页头横条、带筛选框的左侧边栏,或一条竖向导轨,导轨的详细行带有创建时间与状态时长并按活动状态排序;手机首页和桌面首页导轨用的是同一套顺序
|
||||
- **终端外观** —— 七套皮肤(其中四套浅色)、按设备保存的字体与字重(内置的 JetBrains Mono 覆盖 100 到 800 的字重),以及可选启用的入场动画,覆盖标签、智能体窗口、终端面板和连接线
|
||||
- **操作系统通知与主机名感知标题** —— 桌面提醒与标签标题以 `codeman:<host>` 为前缀,使多主机配置不再含糊
|
||||
|
||||
---
|
||||
@@ -416,7 +470,8 @@ PTY 输出 → 16ms 服务端批处理 → DEC 2026 包裹 → SSE → 客户端
|
||||
- **资源模板** —— 展开复选框可选 **Small / Medium / Large / GPU** 预设(内存、CPU、GPU),也可以完全自定义。**磁盘是弹性的** —— 存储随数据增长,没有固定上限。
|
||||
- **按案例共享容器** —— 多个会话可以 `docker exec` 进同一个容器;结束某个会话绝不会影响其他会话所在的容器。
|
||||
- **默认加固** —— 非 root、`--cap-drop ALL`、`no-new-privileges`、PID/内存上限,绝不使用 `--privileged` 或 docker socket;**密封(sealed)** 配置(不注入主机凭据、关闭网络)只需一个开关。
|
||||
- **无感认证、凭据隔离** —— 主机上的 Claude / Codex / Antigravity / Gemini / OpenCode / Pi 登录在容器内开箱即用:凭据在启动时以只读种子方式复制注入,onboarding/信任提示已预先答复,不会弹出登录向导。容器保留自己的副本,绝不回写主机的凭据存储;跨边界共享的只有对话转录,导出文件也绝不包含机密。
|
||||
- **无感认证、凭据隔离** —— 主机上的 Claude / Codex / Antigravity / Gemini / OpenCode / Pi / Grok / OMP 登录在容器内开箱即用:凭据在启动时以只读种子方式复制注入,onboarding/信任提示已预先答复,不会弹出登录向导。容器保留自己的副本,绝不回写主机的凭据存储;跨边界共享的只有对话转录,导出文件也绝不包含机密。
|
||||
- **挂到你已经在跑的容器上** —— 在 Docker 面板勾选 **Attach to an existing container**,就能把 case 链接到一个现成容器,而不是新建一个。Codeman 只 `exec` 进去,绝不启动、停止、重启或删除它;一个被接管的容器可以在不同目录下支撑多个 case,**复制一个已有 case** 会用同一容器上的兄弟 case 预填表单。多用户模式下仅管理员可用,因为容器的挂载属于启动它的人。
|
||||
- **迁移到另一台机器** —— 把容器的完整环境(工具链 + 工作区)导出为可移植的 `.tar.gz`,在另一台机器上导入到新案例即可继续。
|
||||
- **持久耐用** —— Codeman 重启后重连会回到同一个存活的智能体;容器停止/重启后则从绑定挂载的转录恢复对话。
|
||||
|
||||
@@ -433,6 +488,7 @@ PTY 输出 → 16ms 服务端批处理 → DEC 2026 包裹 → SSE → 客户端
|
||||
- **发现与附着**:列出主机上已在运行的 `codeman-*` 会话(由那台机器自己的 Codeman 或其他操作者启动)并附着其一。非你所有的已附着会话在关闭标签时**只分离,绝不杀掉**。
|
||||
- **共享会话**:多个客户端可以以不同窗口尺寸同时附着同一个远程会话而互不挤压;发现列表会显示带客户端计数的「shared」徽标。
|
||||
- **注入安全**:所有 ssh 命令行都经由单一的 shell 转义构建器生成,主机/路径/身份文件字段均有模式校验。
|
||||
- **文件也行**:远程 case 里的预览、下载和文本读取走同一条 ssh 连接(一次 `realpath` + `stat` 探测,然后流式 `cat`,支持 `Range` 拖动进度),所以点一个路径打开的就是智能体所在那台机器上的文件。什么都不会复制到 Codeman 主机;编辑和 Office 预览会明确返回 400,而不是一个误导性的 404。
|
||||
|
||||
在 **New Case → Remote** 中配置(主机、用户、身份文件、可选跳板机)。完整设计:[`docs/remote-sessions.md`](docs/remote-sessions.md)。
|
||||
|
||||
@@ -486,7 +542,7 @@ codeman users list
|
||||
systemctl --user enable codeman-tunnel
|
||||
loginctl enable-linger $USER
|
||||
|
||||
# 或通过 Codeman Web UI:Settings → Tunnel → 切换为开
|
||||
# 或通过 Codeman Web UI:App Settings → System → Remote access → Cloudflare Tunnel
|
||||
```
|
||||
|
||||
</details>
|
||||
@@ -588,7 +644,7 @@ Codeman 默认用 `--dangerously-skip-permissions` 启动会话,因此 Web UI
|
||||
- **默认仅环回** —— 绑定 `127.0.0.1`,仅可从本机访问,因此「无密码」默认配置开箱即安全。在未设置 `CODEMAN_PASSWORD` 的情况下绑定非环回主机会*启动但打印一条醒目警告*,并给出三个具体修复方案(设置密码、环回 + 一个带认证的隧道,或用 `--allow-unauthenticated-network` 显式确认)
|
||||
- **可选认证,真实会话** —— 通过 `CODEMAN_USERNAME`(默认 `admin`)/ `CODEMAN_PASSWORD` 的 HTTP Basic 认证。成功后签发一个不透明的 256 位 `codeman_session` cookie(`randomBytes(32)`)—— 服务端校验,而非客户端签名,因此无法离线伪造(24h TTL、自动延长、设备上下文审计日志)
|
||||
- **按 IP 速率限制** —— 失败 10 次 → `429` 并带 `Retry-After`(15 分钟衰减)。即便攻击者在同一 IP 上猛攻,有效 cookie 或正确密码也能*立即*恢复 —— 这很重要,因为所有隧道流量共享同一个环回 IP。二维码认证有自己独立的限制器
|
||||
- **可配置的权限模式**:`--dangerously-skip-permissions` 只是默认值。**App Settings → Claude CLI → Startup Mode** 可以把新会话切换为 Anthropic 的分类器护栏 `auto` 模式(低打扰,需要 Claude Code 2.1.207+)、`normal` 提示模式,或一份显式的允许工具列表。多用户模式下,未获授权的用户会被强制为 `auto`,shell 会话与跳过权限需要按用户显式授权
|
||||
- **可配置的权限模式**:`--dangerously-skip-permissions` 只是默认值。**App Settings → Agents & CLIs → Claude → Startup Mode** 可以把新会话切换为 Anthropic 的分类器护栏 `auto` 模式(低打扰,需要 Claude Code 2.1.207+)、`normal` 提示模式,或一份显式的允许工具列表。多用户模式下,未获授权的用户会被强制为 `auto`,shell 会话与跳过权限需要按用户显式授权
|
||||
|
||||
### 始终开启的浏览器加固(v0.9.5)
|
||||
|
||||
@@ -602,8 +658,8 @@ Codeman 默认用 `--dangerously-skip-permissions` 启动会话,因此 Web UI
|
||||
|
||||
### 输入、文件与响应头
|
||||
|
||||
- **模式校验的输入** —— 每个 API 请求体都用 Zod v4 模式检查;一个 `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `ANTIGRAVITY_*` / `GEMINI_*` / `GOOGLE_*` / `PI_*` 环境变量前缀允许列表把控每个 CLI 能接收哪些设置
|
||||
- **路径限定** —— 文件路由在边界检查前先 `realpath`(无 TOCTOU);`..`、绝对路径、以及解析到工作目录之外的符号链接都会被拒绝。上限:10 MB 文本预览 / 50 MB 原始与下载;`/api/download` 对敏感路径(`.env`、`*credentials*`、`~/.ssh/`、`.aws/credentials`)做黑名单。SVG/HTML 以 `octet-stream` + `nosniff` + attachment 提供,因此会被下载而非执行
|
||||
- **模式校验的输入** —— 每个 API 请求体都用 Zod v4 模式检查;一个 `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `ANTIGRAVITY_*` / `GEMINI_*` / `GOOGLE_*` / `PI_*` / `GROK_*` / `XAI_*` / `DSH_*` / `DEEPSEEK_*` / `OMP_*` 环境变量前缀允许列表把控每个 CLI 能接收哪些设置,而那些能把 CLI 流量改道的键(base URL、配置目录)对非管理员用户会被钳制
|
||||
- **路径限定** —— 文件路由在边界检查前先 `realpath`(无 TOCTOU);`..`、绝对路径、以及解析到工作目录之外的符号链接都会被拒绝。上限:10 MB 文本预览 / 2 GB 原始与下载(`CODEMAN_MAX_DOWNLOAD_BYTES`;响应体是流式的并支持 `Range` 请求,所以这个上限只是合理性边界,不是内存保护);`/api/download` 对敏感路径(`.env`、`*credentials*`、`~/.ssh/`、`.aws/credentials`)做黑名单。SVG/HTML 以 `octet-stream` + `nosniff` + attachment 提供,因此会被下载而非执行
|
||||
- **安全响应头** —— `Content-Security-Policy`(`default-src 'self'`,每个例外都逐条列举)、`X-Content-Type-Options: nosniff`、`X-Frame-Options: SAMEORIGIN`、HTTPS 下的 HSTS,以及**仅**对 `localhost` / `127.0.0.1` / `::1` 反射的 CORS
|
||||
|
||||
### 供应链与隔离
|
||||
@@ -615,6 +671,22 @@ Codeman 默认用 `--dangerously-skip-permissions` 启动会话,因此 Web UI
|
||||
|
||||
---
|
||||
|
||||
## 终端界面(`codeman tui`)
|
||||
|
||||
一个在终端里运行的全屏会话仪表盘。状态与 Web UI 完全一致,因为它就是同一个服务器的客户端:
|
||||
|
||||
```bash
|
||||
codeman tui # 仪表盘
|
||||
codeman tui --list # 带编号的会话列表,随即退出(可用于脚本)
|
||||
codeman tui 2 # 直接附着到列表里的第 2 个会话
|
||||
```
|
||||
|
||||
会话按 **NEEDS YOU → WORKING → IDLE → RECENT** 分组,等得最久的排最前。`↑↓`/`j`/`k` 选择,`1`-`9` 与 `[`/`]` 切换会话,`Enter` 附着进 tmux 面板(按 **`F1`** 回来)。在面板里,顶部的横条会一直显示会话条,`Alt+1`-`Alt+9` 不用离开就能切换。`y`/`n`/数字可以直接在列表里回答待处理的权限对话框,`p` 发送一行提示,`n` 新建会话并直接进入,`x` 杀掉一个(`y` 确认),`/` 搜索,`g` 显示离开摘要,`?` 是帮助,`q` 退出。窄于 72 列时它会去掉预览面板、变成单列列表,所以在手机上的 Termius 里依然好用。没有服务器在跑时,它仍会以仅附着的降级模式启动。
|
||||
|
||||
Web UI 仍是主要界面;完整指南见 **[docs/tui.md](docs/tui.md)**(英文)。
|
||||
|
||||
---
|
||||
|
||||
## 键盘快捷键
|
||||
|
||||
> Ctrl 绑定在 macOS 上也接受 Cmd。
|
||||
@@ -626,15 +698,21 @@ Codeman 默认用 `--dangerously-skip-permissions` 启动会话,因此 Web UI
|
||||
| `Ctrl/Cmd+Tab` | 下一个会话 |
|
||||
| `Alt/Option+[` / `Alt/Option+]` | 上一个 / 下一个会话 |
|
||||
| `Alt/Option+1`–`Alt/Option+9` | 切换到第 N 个标签(按物理键位,macOS Option 布局也适用) |
|
||||
| `Alt/Option+B` | 折叠 / 展开会话侧边栏(仅侧边栏布局) |
|
||||
| `Ctrl+Shift+{` / `Ctrl+Shift+}` | 将当前标签左移 / 右移 |
|
||||
| `Ctrl/Cmd+C` | 复制选中内容;未选中时中断代理 |
|
||||
| `Ctrl+Shift+C` | 复制选中内容(永不中断) |
|
||||
| `Ctrl/Cmd+V` | 粘贴,或上传剪贴板里的图片并粘贴其路径 |
|
||||
| `Ctrl/Cmd+L` | 清屏 |
|
||||
| `Ctrl+Shift+R` | 恢复终端尺寸 |
|
||||
| `Ctrl+Shift+V` | 切换语音输入 |
|
||||
| `Ctrl/Cmd +` / `-` | 字体大小 |
|
||||
| `Ctrl/Cmd+?` | 键盘帮助 |
|
||||
| `Shift+Enter` | 插入换行(发送到终端) |
|
||||
| `Shift+拖动` | 在鼠标事件交给 CLI 的面板里选中文本 |
|
||||
| 右键 | 复制选中内容(没有选中时保留原生菜单) |
|
||||
| `Shift+滚轮` | 滚轮被转发给 CLI 时,滚动本地回滚缓冲区 |
|
||||
| `Ctrl+Z` | 在智能体会话里被吞掉,运行中的 CLI 不会被挂起;shell 里照常是作业控制 |
|
||||
| `Escape` | 关闭面板与模态框 |
|
||||
|
||||
---
|
||||
@@ -643,16 +721,78 @@ Codeman 默认用 `--dangerously-skip-permissions` 启动会话,因此 Web UI
|
||||
|
||||
面向不经浏览器控制 Codeman 的 AI 智能体与自动化:一个拉起工作会话的智能体、一个 CI 机器人,或是**运行在 Codeman 会话*内部*、编排其他会话的 Claude Code**。UI 能做的一切都是 HTTP + CLI,因此智能体也能做。
|
||||
|
||||
> **捷径:装上打包好的智能体技能。** 下面这一整套(外加多工作会话的实战配方)已经作为 Claude Code 技能随仓库发布在 [`skills/codeman`](skills/codeman/SKILL.md),会话内部的智能体不必等你把文档粘进提示词就能驱动 Codeman。三种获取方式:
|
||||
>
|
||||
> - `npx skills add Ark0N/Codeman --skill codeman -g`:全局安装,任何支持技能的智能体都能用
|
||||
> - Claude Code 插件:`/plugin marketplace add Ark0N/Codeman`,然后 `/plugin install codeman@codeman`:通过 Claude Code 自带的插件管理器全局安装,`/plugin update codeman` 跟随新版本;与 `codeman skill install` 二选一,两者都装会让技能出现两次(`codeman` 和 `codeman:codeman`)
|
||||
> - `codeman skill install`(全局)或 `codeman skill install --case <name>`:给那些从 npm 安装、从未克隆过仓库的用户;`codeman skill uninstall` 可撤销
|
||||
> - **App Settings → Agent Skill**(`agentSkillEnabled`,默认关闭):开启后,Codeman 会在每次于某个 case 中创建 Claude 会话时把技能注入该 case;case 里用户自己写的 `skills/codeman` 永远不会被覆盖
|
||||
>
|
||||
> 全局安装(`codeman skill install` 或 `npx skills add`)会被**本机每一个新建的 Claude Code 会话**读到,无论它在不在 Codeman 里。技能自带门禁:不在 Codeman 会话中(`CODEMAN_MUX` 未设置)时它拒绝动作,所以全局装上它对无关会话没有代价。
|
||||
>
|
||||
> ⚠️ 把 `agentSkillEnabled` 关回去**不会删掉已经注入的副本**(在创建时做清扫,会把技能从共用同一个 `.claude/` 目录的其他活动会话脚下抽走)。要删就按 case 删:`codeman skill uninstall --case <name>`。
|
||||
### 智能体技能(从这里开始)
|
||||
|
||||
这一节的所有内容也打包成了一个 **Claude Code 技能**,位于 [`skills/codeman`](skills/codeman/SKILL.md)。装一次,就再也不用把 API 文档粘进提示词。你用大白话说想要什么,已经坐在 Codeman 会话里的智能体会自己加载配方并驱动 API。
|
||||
|
||||
#### 第 1 步:安装
|
||||
|
||||
| 方式 | 命令 | 范围 |
|
||||
| ---------------- | ---------------------------------------------------------- | ------------------------------------------------------------------------------------------ |
|
||||
| Skills CLI | `npx skills add Ark0N/Codeman --skill codeman -g` | 全局,任何支持技能的智能体都能用 |
|
||||
| Claude Code 插件 | `/plugin marketplace add Ark0N/Codeman`,然后 `/plugin install codeman@codeman` | 全局,通过 Claude Code 自带的插件管理器;`/plugin update codeman` 跟随新版本。与 `codeman skill install` 二选一:两者都装会让技能出现两次(`codeman` 和 `codeman:codeman`) |
|
||||
| 内置 CLI | `codeman skill install` | 全局(`~/.claude/skills/codeman`),给那些从 npm 安装、从未克隆过仓库的用户 |
|
||||
| 内置 CLI | `codeman skill install --case <name>` | 仅一个 case |
|
||||
| Web UI | App Settings → Agents & CLIs → Claude → **Agent Skill** | 每次在某个 case 创建 Claude 会话时自动注入(`agentSkillEnabled`,跨设备同步,默认关闭) |
|
||||
|
||||
`codeman skill uninstall [--case <name>]` 可以撤销 CLI 安装,并且绝不会碰你自己写的 `skills/codeman`。
|
||||
|
||||
#### 第 2 步:开口要
|
||||
|
||||
整个界面就这么多。不用 curl,不用端点名,不用会话 id。下面这些提示照原样就能用:
|
||||
|
||||
| 你说 | 技能做的事 |
|
||||
| ------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------- |
|
||||
| _「现在有哪些会话在跑?」_ | 列出它们的名字、模式和状态。只读,随时可以问。 |
|
||||
| _「在 `myapp` case 上起一个 shell 工作会话,跑测试套件,告诉我过没过。」_ | 拉起、等待一个拆开的完成标记、读回退出码、清理。 |
|
||||
| _「起 3 个工作会话分别跑 lint、typecheck 和测试。并行跑,报告失败的。」_ | 扇出流程:每个任务一个会话,先全部启动,再逐个收集完成的。 |
|
||||
| _「让一个 claude 工作会话在 `refactor-auth` 上总结 `src/session.ts`,然后关掉它。」_ | 拉起、走完就绪阶梯(包括首次运行的信任对话框)、发送并等待、读取干净的 transcript 答案、删除。 |
|
||||
| _「盯着会话 w4,如果它卡在权限提示上就告诉我。」_ | 阻塞在 `blocked` 信号上,并把问题交给**你**。它绝不会替另一个会话回答提示。 |
|
||||
|
||||
#### 第 3 步:没有了
|
||||
|
||||
智能体会删掉它启动的每一个会话。你可以在仪表盘里看着标签出现又消失。
|
||||
|
||||
#### 一次真实的运行,从头到尾
|
||||
|
||||
> **你:** 起 3 个 shell 工作会话,并行跑 lint / typecheck / 前端语法检查,告诉我哪个失败了。
|
||||
|
||||
```text
|
||||
lint -> 9f2d8e5f dispatched
|
||||
typecheck -> aff9c691 dispatched 仪表盘里出现 3 个标签
|
||||
syntax -> be9f1f15 dispatched
|
||||
|
||||
lint DONE_lint_17909 rc=0
|
||||
typecheck DONE_typecheck_3409 rc=0 每完成一个就收集一个
|
||||
syntax DONE_syntax_18501 rc=0
|
||||
|
||||
deleted 9f2d8e5f, aff9c691, be9f1f15 标签消失
|
||||
```
|
||||
|
||||
那些 `DONE_<task>_<random>` 字符串就是技能的**拆分标记**技巧,也是扇出在没有 hook 的 `shell` 会话上依然可靠的原因:敲进去的那一行只含 `${M}_17909`,因此只有命令真正的*输出*里才会出现 `DONE_17909`。不拆开的标记会在命令还没跑之前就匹配到你自己按键的回显。
|
||||
|
||||
#### 盒子里有什么
|
||||
|
||||
| 文件 | 内容 |
|
||||
| ------------------------------------------------------------------- | ------------------------------------------------------------------------------------------- |
|
||||
| [`SKILL.md`](skills/codeman/SKILL.md) | 安全规则、现成的快速路径(起 N 个工作会话、派任务、收集)和动词索引。始终加载。 |
|
||||
| [`reference/verbs.md`](skills/codeman/reference/verbs.md) | 14 个动词的详细说明:就绪、发送并等待、标记、中断、清理。按需加载。 |
|
||||
| [`reference/recipes.md`](skills/codeman/reference/recipes.md) | 8 个完整流程:claude、DeepSeek Harness 与 shell 工作会话、扇出、盯住被卡住的工作会话、消息扇出。按需加载。 |
|
||||
| [`reference/endpoints.md`](skills/codeman/reference/endpoints.md) | 完整端点表、错误码、各模式的信号表、容量限制。按需加载。 |
|
||||
| [`reference/messaging.md`](skills/codeman/reference/messaging.md) | 通过 Claude Code 跨会话消息直接和 claude 工作会话对话。按需加载。 |
|
||||
|
||||
里面的每一个配方都在真实服务器上验证过,注释记录的是实测出来而不是猜出来的失败模式。
|
||||
|
||||
#### 两件值得知道的事
|
||||
|
||||
- **它会自我门禁。** 不在 Codeman 会话里(`CODEMAN_MUX` 未设置)时,技能拒绝动作,也不去猜 API 地址,所以全局安装对无关的 Claude Code 会话没有任何代价。
|
||||
- **它刻意保守。** 未经提示,它只会拉起会话、给它们发提示,并删除**它在同一段对话里自己创建的**会话(按精确 id,经由一个拒绝删除智能体自身会话的失败即关闭守卫)。删除 case(会抹掉一个真实的代码目录)、批量杀会话、改动 respawn/ralph/cron/orchestrator 以及写设置,都需要你开口并指名目标。
|
||||
|
||||
⚠️ 把 `agentSkillEnabled` 关回去**不会删掉已经注入的副本**(在创建时做清扫,会把技能从共用同一个 `.claude/` 目录的其他活动会话脚下抽走)。要删就按 case 删:`codeman skill uninstall --case <name>`。
|
||||
|
||||
---
|
||||
|
||||
**这一节余下的部分是手动路径**:同样的操作用裸 HTTP 来做,适合 CI 机器人、shell 脚本,或任何不支持技能的智能体。
|
||||
|
||||
### 检测自己身处 Codeman 内部
|
||||
|
||||
@@ -673,7 +813,7 @@ Codeman 默认用 `--dangerously-skip-permissions` 启动会话,因此 Web UI
|
||||
4. **响应信封。** 多数端点返回 `{ "success": true, "data": … }`(错误:`{ "success": false, "error", "errorCode" }`)。少数遗留 GET 返回裸响应体 —— **两种都要处理**(`body.data ?? body`)。
|
||||
5. **`/api/v1/*`** 是 `/api/*` 的稳定别名。
|
||||
6. **用等待代替轮询,别把超时当成错误。** 等待类端点在没等到事情发生时也以 HTTP `200` 加 `wait.timedOut: true` 应答,所以要循环调用短等待(默认 60 秒),而不是发一个超长的调用:隧道会掐断空闲连接。`wait.timeoutMs` 告诉你服务端钳制之后真正采用的超时(上限 600 秒)。
|
||||
7. **只有 `claude` 会话会发出 `stop` 与 `blocked`。** 这两个来自 Claude Code hook;`shell` 与外部 CLI(opencode/codex/gemini/antigravity/pi)只接受 `idle`、`working` 与 `exit`。在这些模式上显式索要 `stop` 会得到 `400`;不传 `until` 则永远安全。⚠️ `shell` 会话的 `idle` 只在启动时触发**一次**,此后再也不会,所以在那里用「发送并等待」只能等到超时:没有 hook 的会话请用 `wait-output` 标记来同步。
|
||||
7. **只有 `claude` 与 `deepseek` 会话会发出 `stop` 与 `blocked`。** 这两个来自 hook(Claude Code 自己的,以及 DeepSeek Harness 的状态桥接);`shell` 与其他外部 CLI(opencode/codex/gemini/antigravity/pi/grok/omp)只接受 `idle`、`working` 与 `exit`。在这些模式上显式索要 `stop` 会得到 `400`;不传 `until` 则永远安全。⚠️ `shell` 会话的 `idle` 只在启动时触发**一次**,此后再也不会,所以在那里用「发送并等待」只能等到超时:没有 hook 的会话请用 `wait-output` 标记来同步。
|
||||
8. **没有任何东西会报告「就绪」,得自己显式等。** 新会话在 PID 出现之前一律回答 `{"signal":"exit","immediate":true}`(意思是*还没启动*,不是*崩了*),而全新 case 里的 `claude` 工作会话接着会停在 CLI 的信任对话框上。此时给它发提示,等待会在约 2 秒后因 `idle` 解除,看上去和一个跑完的回合一模一样,而文本其实卡在对话框里。下面的配方 2b 就是避开它的顺序。
|
||||
|
||||
### 常用配方
|
||||
@@ -738,7 +878,7 @@ curl -sG "$API/api/sessions/$SID/wait-output" \
|
||||
--data-urlencode "match=DONE_$N" --data-urlencode 'from=buffer' \
|
||||
--data-urlencode 'timeout=60000' | jq '.data.wait'
|
||||
|
||||
# 5. 读回答案。claude / codex 会话用 last-response:它取自 transcript 而不是屏幕,
|
||||
# 5. 读回答案。claude / codex / deepseek 会话用 last-response:它取自 transcript 而不是屏幕,
|
||||
# 因此不带 TUI 的画框与重画噪声。⚠️ 要轮询,别只读一次:transcript 落盘比 stop
|
||||
# 信号稍晚,紧跟着「发送并等待」返回后立刻读,常常拿到空串。
|
||||
for _ in $(seq 1 10); do
|
||||
@@ -747,7 +887,7 @@ for _ in $(seq 1 10); do
|
||||
done
|
||||
printf '%s\n' "$TXT"
|
||||
|
||||
# 5b. 其他模式(shell/opencode/gemini/antigravity/pi)没有 transcript,读终端。
|
||||
# 5b. 其他模式(shell/opencode/gemini/antigravity/pi/grok/omp)没有 transcript,读终端。
|
||||
# ⚠️ 用 terminal?tail=,不要用 /output:后者的 textOutput 对每个由 tmux 承载的
|
||||
# (也就是每个交互式)会话都是空的。tail 按字节计,返回的是含 ANSI 的终端数据。
|
||||
curl -s "$API/api/sessions/$SID/terminal?tail=8000" | jq -r '.data.terminalBuffer'
|
||||
@@ -780,7 +920,9 @@ codeman session start -d /path/to/repo # (s) 启动会话
|
||||
codeman session list # 列出会话
|
||||
codeman session logs <id> # 查看输出
|
||||
codeman task add "fix the failing test" # (t) 排入任务
|
||||
codeman attach <path> # 附着 Claude hook 上下文
|
||||
codeman attach <path> # 为本地文件显示一张附件卡片
|
||||
codeman tui --list # 带编号的会话列表(管道输出时为纯文本)
|
||||
codeman tui 3 # 附着到该列表里的第 3 个会话
|
||||
```
|
||||
|
||||
### Hook(事件*回流*到 Codeman)
|
||||
@@ -793,7 +935,7 @@ Codeman 会注册 Claude Code hook,它们 `POST /api/hook-event`(`permission
|
||||
|
||||
## API
|
||||
|
||||
基于 Fastify 的 REST —— **21 个路由模块中约 200 个处理器**,外加一条 SSE 流和一条 WebSocket 终端通道。所有响应都使用 `ApiResponse<T>` 信封(`{success, data}` / `{success, error, errorCode}`);`/api/v1/*` 是稳定别名。以下是一个有代表性的子集:
|
||||
基于 Fastify 的 REST —— **25 个路由模块中约 230 个处理器**,外加一条 SSE 流和一条 WebSocket 终端通道。所有响应都使用 `ApiResponse<T>` 信封(`{success, data}` / `{success, error, errorCode}`);`/api/v1/*` 是稳定别名。以下是一个有代表性的子集:
|
||||
|
||||
### 会话(Sessions)
|
||||
|
||||
@@ -804,11 +946,13 @@ Codeman 会注册 Claude Code hook,它们 `POST /api/hook-event`(`permission
|
||||
| `POST` | `/api/sessions/:id/input` | 发送输入(`{input, useMux?, clientId?, seq?, wait?, waitTimeout?}`:`clientId`+`seq` = 精确一次;`wait` 阻塞到这一回合结束) |
|
||||
| `GET` | `/api/sessions/:id/terminal` | 读取终端输出(`?tail=<bytes>`、`?full=1`):交互式会话的读取路径 |
|
||||
| `GET` | `/api/sessions/:id/output` | 一次性的解析输出(tmux 承载的会话里 `textOutput` 为空) |
|
||||
| `GET` | `/api/sessions/:id/last-response` | 从 transcript 读出的最后一条回答,纯文本(claude、codex、deepseek) |
|
||||
| `GET` | `/api/sessions/:id/wait` | 阻塞到某个信号触发(`?until=stop,idle,exit&timeout=&fresh=`);超时是 `200` |
|
||||
| `GET` | `/api/sessions/:id/wait-output` | 阻塞到某个字面串出现(`?match=&nocase=&from=now\|buffer&timeout=`) |
|
||||
| `GET` | `/api/sessions/unified` | 统一的活动 + 历史清单(会话管理器):`?q=&limit=` |
|
||||
| `POST` | `/api/sessions/:id/pin` | 在会话管理器中置顶 / 取消置顶(`{pinned}`) |
|
||||
| `PUT` | `/api/session-order` | 跨设备同步标签顺序(`{order: [ids]}`) |
|
||||
| `POST` | `/api/sessions/:id/custom-model` | 让会话的 CLI 在一个已保存的自定义端点上原地重启(`{endpointId, modelId}`;`{clear: true}` 回到官方后端) |
|
||||
| `DELETE` | `/api/sessions/:id` | 删除会话 |
|
||||
|
||||
### 重生(Respawn)
|
||||
@@ -857,6 +1001,7 @@ Codeman 会注册 Claude Code hook,它们 `POST /api/hook-event`(`permission
|
||||
| `GET` | `/api/system/update/check` | 检查新发行版 |
|
||||
| `POST` | `/api/system/update` | 自更新(git-clone 安装) |
|
||||
| `POST` | `/api/clipboard` | 把文本推送到所有已连接浏览器(`{text}`) |
|
||||
| `GET` / `POST` | `/api/model-endpoints` | 列出 / 保存自定义的 OpenAI 兼容端点(`PUT` / `DELETE` `/:id`;多用户模式下仅管理员) |
|
||||
| `GET` | `/api/sessions/:id/run-summary` | 时间线 + 统计 |
|
||||
|
||||
> **想在 Codeman 之上做集成?**[`docs/extending-codeman.md`](docs/extending-codeman.md)(英文)是集成指南:把你自己的界面作为标签页嵌入、订阅 SSE 事件流以便在 agent 需要你时做出响应、用脚本驱动 Codeman,以及动手前值得先了解的那些坑。Codeman 刻意不提供插件运行时,所以一个集成就是你自己的进程在讲 HTTP。
|
||||
@@ -893,7 +1038,7 @@ flowchart TB
|
||||
end
|
||||
|
||||
subgraph External["外部"]
|
||||
CLI["AI CLI<br/><small>Claude Code / OpenCode / Codex / Antigravity / Gemini / Pi</small>"]
|
||||
CLI["AI CLI<br/><small>Claude Code / OpenCode / Codex / Antigravity / Gemini / Pi / Grok / DeepSeek / OMP</small>"]
|
||||
BG["后台智能体<br/><small>(Task 工具)</small>"]
|
||||
end
|
||||
end
|
||||
@@ -931,6 +1076,12 @@ npm test # 运行测试(与 CI 相同;浏览器/移动端
|
||||
|
||||
---
|
||||
|
||||
## 社区
|
||||
|
||||
提问、安装求助和想法都在 [GitHub Discussions](https://github.com/Ark0N/Codeman/discussions):[Q&A 板块](https://github.com/Ark0N/Codeman/discussions/categories/q-a)回答了最常见的那些(手机访问、通宵运行、更新),路线图则在 [Ideas](https://github.com/Ark0N/Codeman/discussions/categories/ideas) 里决定。Bug 请提到 [issues](https://github.com/Ark0N/Codeman/issues);报告通常一天内会得到回复,每个发行版都会点名感谢报告者和贡献者。想参与贡献?[CONTRIBUTING.md](.github/CONTRIBUTING.md) 是地图:皮肤、翻译和文档都是很好的第一个 PR,更大的特性先从一个 Discussion 开始。如果你对自己的配置很自豪,发到 [Show and tell](https://github.com/Ark0N/Codeman/discussions/300) 来。
|
||||
|
||||
---
|
||||
|
||||
## 代码库质量
|
||||
|
||||
本代码库经历了一次全面的 7 阶段重构,消除了上帝对象、集中了配置,并建立了模块化架构:
|
||||
@@ -954,7 +1105,7 @@ npm test # 运行测试(与 CI 相同;浏览器/移动端
|
||||
|
||||
[](https://www.npmjs.com/package/xterm-zerolag-input)
|
||||
|
||||
为 xterm.js 提供即时按键反馈的叠加层。通过把输入的字符立即渲染为像素级精准的 DOM 叠加层,消除高 RTT 连接下的感知输入延迟。零依赖、可配置的提示符检测、带 78 个测试的完整状态机。
|
||||
为 xterm.js 提供即时按键反馈的叠加层。通过把输入的字符立即渲染为像素级精准的 DOM 叠加层,消除高 RTT 连接下的感知输入延迟。零依赖、gzip 后 6.1 kB、可配置的提示符检测、CJK/emoji 宽字符支持、带 238 个测试的完整状态机。
|
||||
|
||||
```bash
|
||||
npm install xterm-zerolag-input
|
||||
@@ -977,3 +1128,8 @@ MIT —— 见 [LICENSE](LICENSE)
|
||||
<p align="center">
|
||||
<strong>跟踪会话。可视化智能体。掌控重生。让它在你睡觉时持续运行。</strong>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
如果 Codeman 帮你省了时间,<a href="https://github.com/Ark0N/Codeman/stargazers">点个 star</a> 能让更多人找到它。<br>
|
||||
欢迎到 <a href="https://github.com/Ark0N/Codeman/issues">Issues</a> 报告 bug 和提出特性想法。
|
||||
</p>
|
||||
|
||||
@@ -27,6 +27,8 @@ export const BROWSER_TEST_GLOBS = [
|
||||
'test/webgl-fallback.test.ts',
|
||||
'test/terminal-copy-shortcut.test.ts',
|
||||
'test/terminal-keycode229-recovery.browser.test.ts',
|
||||
'test/capture-load-window.browser.test.ts',
|
||||
'test/capture-geometry-retry.browser.test.ts',
|
||||
'test/codex-predictive-echo.test.ts', // also needs a real codex binary
|
||||
];
|
||||
|
||||
|
||||
@@ -324,6 +324,30 @@ from the session's current state rather than requiring a new transition: the
|
||||
original turn may be long over. It comes back as
|
||||
`"delivered": false, "duplicate": true`.
|
||||
|
||||
**Wake-on-LAN hosts** (`docs/remote-sessions.md` §Wake-on-LAN): when the session's
|
||||
remote host has a wake target and is asleep, the non-wait form answers `200` with
|
||||
`{"buffered": true}` — the bytes are held and flushed after the host is back — or
|
||||
`{"buffered": true, "dropped": true}` for a chunk over the 4 KB wake buffer, which
|
||||
is gone (never delivered as a fragment). Both fields are additive to the historical
|
||||
bare `{}`. With `wait`, the route blocks on the wake instead and answers
|
||||
`422 OPERATION_FAILED` ("did not come back after a wake-on-LAN request — nothing was
|
||||
sent") when the host never returns, rather than writing into the stalled pane and
|
||||
reporting `delivered:true` plus a timeout.
|
||||
|
||||
Two endpoints back that flow directly, both scoped to one session's remote host and
|
||||
both refusing a session that is not remote (`400 INVALID_INPUT`):
|
||||
|
||||
| Method | Path | Purpose |
|
||||
| --- | --- | --- |
|
||||
| `GET` | `/api/sessions/:id/reachability` | Whether the session's remote host answers SSH right now, plus whether a wake target is configured. Read-only: it never wakes. `{"reachable": true\|false\|null, "wakeConfigured": "mac"\|"command"\|"none"}`, where `null` means the answer is unknown (a proxied host, where a TCP probe proves nothing). |
|
||||
| `POST` | `/api/sessions/:id/wake` | Wake the host and wait for it to accept SSH again, bounded by the request budget. `422 OPERATION_FAILED` when it does not come back; `400 INVALID_INPUT` with "No wake-on-LAN target configured for this host" when nothing is set. |
|
||||
|
||||
⚠️ Waking is deliberately reachable only from an explicit user action (this route, a
|
||||
session create/attach, or typing into a sleeping session). No watcher, dropped-session
|
||||
handler or boot-recovery path may wake a host, or a suspended machine would be woken
|
||||
again seconds after every suspend; `test/remote-wake.test.ts` pins that as an import
|
||||
fence around `src/remote-wake.ts`.
|
||||
|
||||
### Response
|
||||
|
||||
All three nest the wait result under `data.wait`, so one client helper works against
|
||||
@@ -479,6 +503,48 @@ re-captured, or the item acknowledged), `approval:resolved` (`{ id, sessionId, k
|
||||
`resolution` one of `answered | resolved_in_terminal | superseded |
|
||||
session_ended | dismissed | expired`).
|
||||
|
||||
## Reboot restore
|
||||
|
||||
A host reboot takes the tmux server down with it, so every pane dies and the
|
||||
board comes up empty. At boot Codeman works out which sessions the reboot
|
||||
destroyed and holds that plan in memory, and these endpoints let a client offer
|
||||
it to the user. Nothing creates a pane until the user asks: the boot-time reboot
|
||||
heuristic decides whether to ASK, never whether to act.
|
||||
|
||||
Claude-mode sessions only (others carry their conversation id in their own
|
||||
config object); remote and docker sessions are never offered, because both need
|
||||
another host or container to be up. The plan is in-memory, so a server restart
|
||||
drops it and the offer is gone; the conversations themselves are unaffected,
|
||||
since they live in the CLI's own transcript store and stay reachable from the
|
||||
Resume list. A plan nobody spends expires after 24 hours.
|
||||
|
||||
- `GET /api/v1/reboot-restore` → `{ sessions: RestorableSession[],
|
||||
scrollbackRestored: false }`, ownership-scoped in multi-user mode.
|
||||
`RestorableSession`: `{ id, name?, workingDir, mode, owner? }`. The persisted
|
||||
record itself is never sent. `scrollbackRestored` is always `false` and exists
|
||||
so a client states it: a restored session is a NEW pane, so the conversation
|
||||
continues and the terminal history does not.
|
||||
- `POST /api/v1/reboot-restore/restore` with `{ sessionIds?: string[] }` (omit
|
||||
to restore everything the caller can see) → `{ restored: RestorableSession[],
|
||||
skipped: { sessionId, reason }[] }`. `reason` is one of `workspace-missing`
|
||||
(the directory is gone), `workspace-forbidden` (in multi-user mode it is
|
||||
outside the workspace of the user the session belongs to, re-checked against
|
||||
that owner's current grant rather than the caller's), `already-live` (the conversation is already
|
||||
open, typically resumed by hand from the Resume list), `capacity-reached`
|
||||
(the global or per-user session cap), or `rebuild-failed` (the agent would not
|
||||
start, most often a CLI binary missing from the server's PATH).
|
||||
`409 CONFLICT` when that caller already has a restore running. Entries are
|
||||
removed from the plan before any pane is built, so a double-click cannot put
|
||||
two panes on one conversation; anything that never became a pane goes back on
|
||||
offer, except `already-live`, which cannot stop being true. A restored session
|
||||
comes back attached, idle and disarmed: respawn controllers and Ralph loops
|
||||
are never re-armed automatically.
|
||||
- `POST /api/v1/reboot-restore/dismiss` → `{ dismissed: n }`. Drops the offer
|
||||
for everything the caller can see.
|
||||
|
||||
Each rebuilt session also emits the ordinary `session:created` SSE event, so
|
||||
clients other than the one that clicked pick it up without refetching.
|
||||
|
||||
## Read My Mind intent profiles
|
||||
|
||||
Per-case profiles of what the user is trying to accomplish: user/agent-stated
|
||||
@@ -516,6 +582,115 @@ All four enforce session ownership in multi-user mode; a foreign session id
|
||||
answers `404 NOT_FOUND` (no existence leak), and profiles of two owners of the
|
||||
same directory are distinct by construction.
|
||||
|
||||
## Custom Model Endpoints
|
||||
|
||||
Points a session's harness at a user-configured OpenAI-compatible endpoint —
|
||||
local (llama.cpp, vLLM, DGX Spark) or cloud (Azure AI Foundry, OpenRouter) —
|
||||
instead of its native cloud backend, gated by the opt-in
|
||||
`customModelEndpointsEnabled` setting (default OFF). Endpoints are
|
||||
machine-level infra, like remote/docker hosts: writes are admin-only in
|
||||
multi-user mode. Design: [`custom-model-endpoints-plan.md`](custom-model-endpoints-plan.md);
|
||||
user guide: [`custom-model-endpoints.md`](custom-model-endpoints.md).
|
||||
|
||||
- `GET /api/v1/model-endpoints` -> `CustomModelHost[]`, an unwrapped bare
|
||||
array like every other list route (still riding the standard `{success,
|
||||
data}` envelope on the wire — unwrap it the same way). Answers `[]` for a
|
||||
non-admin in multi-user mode. `apiKey` is never returned; `apiKeySet:
|
||||
boolean` reports whether one is stored, so a client can render "unchanged
|
||||
if left blank" without ever holding the real value.
|
||||
- `POST /api/v1/model-endpoints` with `{ id, label, baseUrl, apiKey?,
|
||||
authStyle?, defaultModelId? }` creates one. `id` must match
|
||||
`^[a-zA-Z0-9_-]+$`; `authStyle` is `bearer` (default) or `api-key`, never
|
||||
both (a real server hung indefinitely when sent both headers on one
|
||||
request); `baseUrl` must be `http(s)`, carry no embedded credentials, and
|
||||
is refused if it points at (or resolves to) a link-local or
|
||||
cloud-metadata address. `409 ALREADY_EXISTS` on a duplicate id.
|
||||
- `PUT /api/v1/model-endpoints/:id` updates one. An **absent** `apiKey`
|
||||
keeps the stored one rather than clearing it — the client never receives
|
||||
the real value to resend deliberately unchanged, so omission is the only
|
||||
way to say "leave it alone"; there is no way to clear a key back to unset
|
||||
this way. `defaultModelId`, when set, must be one of that endpoint's own
|
||||
`models` (`400 INVALID_INPUT` otherwise).
|
||||
- `DELETE /api/v1/model-endpoints/:id` removes one.
|
||||
- `POST /api/v1/model-endpoints/:id/discover-models` fetches the endpoint's
|
||||
own `GET /v1/models` and stores the result as `models`, updating
|
||||
`lastDiscoveredAt`, plus (best-effort, only for a model llama-swap's own
|
||||
response already reports loaded) `modelContextLengths` and `modelSizesGB`.
|
||||
A `defaultModelId` that no longer appears in the fresh list is dropped
|
||||
rather than carried forward invalid. Failures answer `422 OPERATION_FAILED`
|
||||
with the underlying connection error, or a named egress refusal if the
|
||||
resolved address turned out to be blocked. The same refresh also runs
|
||||
automatically for every saved endpoint every 5 minutes in the background
|
||||
(`refreshAllCustomModelHosts()`, `custom-model-routes.ts`, started from
|
||||
`server.ts`), so there is no route for triggering "refresh all" — one
|
||||
endpoint being unreachable on a cycle never blocks the others.
|
||||
- `GET /api/v1/model-endpoints/:id/running-status` -> `{ isLlamaSwap,
|
||||
running: [{model, state}], logLine? }`, read-only, no admin gate
|
||||
(any session owner who could already point a session at this endpoint can
|
||||
equally ask what it currently has loaded). `isLlamaSwap` is
|
||||
feature-detected via the endpoint's own `GET /running` — a plain
|
||||
llama.cpp/OpenAI-compatible server has none and always answers `false`.
|
||||
`logLine`, present only when `isLlamaSwap` is true, is the most recent
|
||||
REAL backend `llama-server` process log line (`load_model: ...`,
|
||||
`llama_server: model loaded`, etc.), sourced from the endpoint's own
|
||||
`GET /api/events` SSE stream and filtered to `source: "upstream"` frames
|
||||
only (never llama-swap's own `source: "proxy"` request-access log) — one
|
||||
connection is held open per endpoint and reused across every poller,
|
||||
idle-closed after 30s of nobody asking. This is what the Run-menu
|
||||
picker's loading banner polls once a second while a model is loading.
|
||||
- `POST /api/v1/sessions/:id/custom-model` with `{ endpointId, modelId,
|
||||
confirmed? } | { clear: true }` applies (or clears) the session's
|
||||
selection and **restarts the session's CLI process in place** — every
|
||||
supported harness reads its endpoint config at process start, never per
|
||||
turn, so there is no live hot-swap. (`POST /api/v1/quick-start`'s own
|
||||
`customModel: { endpointId, modelId, confirmed? }` field is the
|
||||
no-restart equivalent for a session that doesn't exist yet — see below.)
|
||||
A Claude session resumes its existing conversation across the restart;
|
||||
pi/omp/grok additionally get a forced `--model`/`-m` value, since for
|
||||
those three the config file alone does not select it. `400 INVALID_INPUT`
|
||||
for a remote (SSH) or Docker session — both restart their agent
|
||||
differently under the hood, and applying to one would report success
|
||||
while changing nothing. Two more responses replace the normal
|
||||
`{customModel, restarted}` shape, neither an error, and neither restarts
|
||||
or creates anything on the first ask. ⚠️ **Each is answered by its OWN
|
||||
flag on the retry, and answering one is not consent to the other**: they
|
||||
are questions about different people, and while they shared a single flag
|
||||
a caller who confirmed the context warning silently agreed to evict
|
||||
another session's model as well. Send `confirmedContext: true` to proceed
|
||||
past the context warning, `confirmedSwap: true` past the swap conflict,
|
||||
and both when both were asked (they accumulate, so the second retry still
|
||||
carries the first answer). The original `confirmed: true` still means
|
||||
BOTH and is still accepted, because it shipped in this feature's
|
||||
HTTP-API-only cut; new callers should send the specific one:
|
||||
- `{requiresConfirmation: true, currentlyLoadedModel, affectedSessions}` —
|
||||
llama.cpp/llama-swap only runs one model at a time, and switching would
|
||||
unload a model another **live session's own selection** is actively
|
||||
using. Never returned for a plain (non-llama-swap) server, and never
|
||||
just because a swap is needed at all — only when it would disrupt
|
||||
someone else.
|
||||
- `{requiresContextWarning: true, modelId, contextLength,
|
||||
minSafeContextTokens}` — Claude Code's own fixed per-turn overhead
|
||||
(system prompt + tool schemas) can exceed a small model's entire
|
||||
discovered context on its own, before any conversation history exists
|
||||
to compact, guaranteeing the very first message fails regardless of
|
||||
`CLAUDE_CODE_MAX_CONTEXT_TOKENS`. Gated on the CLI registry declaring a
|
||||
`contextLengthVar` (claude only today), so it never fires for another
|
||||
harness.
|
||||
- `POST /api/v1/quick-start`'s `customModel: { endpointId, modelId,
|
||||
confirmed?, confirmedContext?, confirmedSwap? }` field (alongside its
|
||||
normal `caseName`/`mode`/etc. body)
|
||||
computes the same injection **before** the session exists and launches
|
||||
directly on the endpoint — no restart, because there was never a
|
||||
native-backend boot to restart away from. Runs the identical checks as
|
||||
the dedicated route above (`requiresConfirmation`/`requiresContextWarning`,
|
||||
same shapes, same per-question `confirmedContext`/`confirmedSwap` retry),
|
||||
and is refused the same way
|
||||
for a remote or Docker case. This is what the Run-menu picker uses for
|
||||
opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP; Claude still uses the
|
||||
dedicated restart route above (its `--resume`-based restart is far less
|
||||
jarring than a full relaunch, and folding it into the one-shot path is
|
||||
separate work — see `docs/custom-model-endpoints-plan.md`).
|
||||
|
||||
## Voice dictation
|
||||
|
||||
Browser dictation transcribed through this server's Claude Code login, i.e. the
|
||||
|
||||
File diff suppressed because one or more lines are too long
@@ -104,17 +104,17 @@ declared capability, never an `if (mode === 'claude')` branch.
|
||||
|
||||
## Per-CLI injection recipes (confidence-ranked)
|
||||
|
||||
| CLI | Mechanism | Confidence |
|
||||
| ------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `claude` | Env vars: `ANTHROPIC_BASE_URL`, `ANTHROPIC_API_KEY`, `ANTHROPIC_DEFAULT_SONNET_MODEL`/`_HAIKU_MODEL`/`_OPUS_MODEL` (all set to the chosen model/deployment name) | **Verified end-to-end** against a real llama-swap server — a real "hello world" reply came back. ⚠️ Non-interactive (`-p`) invocations also fire an async session-title-generation call that reuses `ANTHROPIC_DEFAULT_HAIKU_MODEL` and validates it against Claude Code's OWN internal recognized-model list, printing `[claude-code:unrecognized_model]` and, in `-p` mode, hanging the whole invocation rather than just warning. `--settings '{"autoTitle":false}'` does NOT stop this (confirmed); `--bare` does (the warning still prints, but the real prompt runs) — but `--bare` ALSO disables hooks, LSP, plugin sync, and CLAUDE.md auto-discovery, so it is only safe for the standalone one-shot test script, NEVER for a real interactive Codeman session (which depends on hooks for idle detection, trust-dialog auto-accept, etc. — see the External CLI modes section of CLAUDE.md). Whether an INTERACTIVE claude session with a custom model hits the same hang (vs. just a background warning) is untested and should be checked before calling chunk 5/6 done for claude |
|
||||
| `opencode` | `OPENCODE_CONFIG_CONTENT` env var (already a registry mechanism, `stock.ts:342`) holding a JSON blob: `{"provider":{"custom":{"options":{"baseURL":...,"apiKey":...},"models":{"<name>":{}}}},"model":"custom/<name>"}` | **Verified by user** |
|
||||
| `codex` | TOML `config.toml`: top-level `model = "<id>"` + `[model_providers.custom]` (`base_url`, `env_key` naming an env var the real API key rides in — never a literal TOML field, since codex's schema has no such field). Written to an isolated dir via `CODEX_HOME` (`stock.ts:405-415`) so the user's own `~/.codex/config.toml` is never touched | **Config STRUCTURE verified** against a real codex binary (an earlier `[model].default` table shape was rejected: "invalid type: map, expected a string" — caught live). **Protocol CONFIRMED BROKEN against llama.cpp/llama-swap**: codex only speaks the Responses API (`wire_api = "responses"`, the only value it accepts since it dropped `"chat"` support in Feb 2026), and a real llama-swap server does not implement `/v1/responses` — a live run against it failed with repeated `Reconnecting...` then `high demand` errors. Codex support therefore needs a Responses-API-compatible endpoint (most local llama.cpp/Ollama/vLLM setups do not qualify); do not present this as working against a generic OpenAI-Chat-Completions box |
|
||||
| `gemini` | Env vars `GOOGLE_GEMINI_BASE_URL` + `GEMINI_API_KEY` + `GEMINI_MODEL`; CLI needs a restart to pick them up | **Confirmed BROKEN against llama.cpp/llama-swap, unresolved after real investigation.** Setting `GOOGLE_GEMINI_BASE_URL` makes gemini-cli internally select an `AuthType.GATEWAY` auth path (undocumented — inferred from behaviour) with validation requirements distinct from every normal auth mode; a real run against llama-swap fails with `Invalid auth method selected` regardless of what key/format is supplied. Tried and all failed: a Google-format dummy API key, `GOOGLE_GENAI_USE_VERTEXAI=false`, a `GEMINI_DEFAULT_AUTH_TYPE` override, and hand-writing `settings.json` directly. `--skip-trust` was a real, separate fix (without it a trust-folder check silently overrides `--approval-mode yolo` back to `default`) but does not touch this auth failure. Documented as an open gap, not shipped as working — the registry entry and injection code exist and are exercised by the test script, but end-to-end gemini support needs upstream investigation of `GATEWAY` AuthType before it can be called done |
|
||||
| `pi` | Config file `~/.pi/agent/models.json` with a custom provider whose `models` is an **array** of `{id}` objects (not an object keyed by id) plus `authHeader: true`. Redirected via the child process's own `HOME` env var, isolated per test/session — **not** `PI_CONFIG_DIR`, which does nothing for pi (grepped pi's entire bundled JS source: the string appears nowhere) | **Verified end-to-end** against a real llama-swap server — real "hello world" reply came back. Two real bugs found and fixed before this worked: (1) `PI_CONFIG_DIR` is not read by pi at all — pi hardcodes `~/.pi/agent/models.json` with no dedicated override, so the actual redirect has to be the child process's `HOME`; (2) `models` must be an array of `{id}` objects per pi's own bundled `docs/models.md`, not an object keyed by model id (silently loaded zero models). Also requires an explicit `--model custom/<id>` on invocation — without it pi falls back to its own default provider and fails with "No API key found for the selected model" |
|
||||
| `grok` | TOML `config.toml`: a fixed `[model.codeman-custom]` block (`base_url`, `env_key` naming an env var the key rides in, never a literal TOML field) written to an isolated dir via `GROK_HOME`. Invoked with `-m codeman-custom` | **Verified end-to-end** against a real llama-swap server — real "hello world" reply came back. The ORIGINAL recipe in this table (env vars `GROK_BASE_URL`/`XAI_API_KEY`/`GROK_MODEL`) was flat-out **wrong**, not just unverified: it produced "Not signed in" against a real binary. Grok's real mechanism, confirmed against xAI's own docs and a live binary, is a `config.toml` with a `[model.<name>]` block, redirected via `GROK_HOME`; the key still rides as an env var (`XAI_API_KEY` via `env_key`), just referenced from the TOML rather than read directly |
|
||||
| `deepseek` | Reuse the **existing** `DEEPSEEK_BASE_URL` + `DEEPSEEK_API_KEY` keys (already declared in `stock.ts`). Only `DEEPSEEK_BASE_URL` is in `privilegedEnvKeys` — `DEEPSEEK_API_KEY` deliberately stays clamp-exempt, since a non-granted owner supplying their OWN key removes privilege rather than granting it (adding it to the clamp list was a real regression, caught by `test/deepseek-mode.test.ts` and fixed before merge). No model-selection var — dsh model is a profile composition entry, not a flag/env var | **Confirmed reaching the server, but failing — unresolved.** A real run against llama-swap returns `dsh: HTTP_404: DeepSeek API error (HTTP 404)` consistently (confirmed the env vars are read: the request reaches the network rather than failing locally). Root cause not identified — plausible explanation by analogy with codex's Responses-API gap is that `dsh --profile headless` expects DeepSeek's official API response shape/path structure rather than a generic OpenAI-compatible `/v1/chat/completions` endpoint, but this was not confirmed by reading dsh's own bundled source (unlike pi/grok, where that grep resolved the question directly). Documented as best-effort/unknown, not shipped as verified working |
|
||||
| `omp` | Config file `~/.omp/agent/models.yml` with the same array-shaped `models` + `authHeader: true` fix as pi. Redirected via `HOME`, same reasoning as pi (`PI_CONFIG_DIR` does not relocate omp's config either, despite an earlier CLAUDE.md note claiming it does) | **Verified end-to-end** against a real llama-swap server — real "hello world" reply came back, after applying the same two fixes as pi (array-shaped `models`, `HOME`-redirect instead of `PI_CONFIG_DIR`) plus an explicit `--model custom/<id>` on invocation. Unverified against omp's own official docs (none are bundled in the install), but empirically confirmed working live |
|
||||
| `antigravity` | No CLI/env/config mechanism found — Antigravity's docs describe only a GUI settings panel, and explicitly say a custom endpoint "cannot currently" become the core reasoning model. **Not implemented**; toolbar entry stays disabled for this mode with an explanatory tooltip | No known mechanism |
|
||||
| CLI | Mechanism | Confidence |
|
||||
| ------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `claude` | Env vars: `ANTHROPIC_BASE_URL`, `ANTHROPIC_API_KEY`, `ANTHROPIC_DEFAULT_SONNET_MODEL`/`_HAIKU_MODEL`/`_OPUS_MODEL` (all set to the chosen model/deployment name) | **Verified end-to-end** against a real llama-swap server — a real "hello world" reply came back. ⚠️ Non-interactive (`-p`) invocations also fire an async session-title-generation call that reuses `ANTHROPIC_DEFAULT_HAIKU_MODEL` and validates it against Claude Code's OWN internal recognized-model list, printing `[claude-code:unrecognized_model]` and, in `-p` mode, hanging the whole invocation rather than just warning. `--settings '{"autoTitle":false}'` does NOT stop this (confirmed); `--bare` does (the warning still prints, but the real prompt runs) — but `--bare` ALSO disables hooks, LSP, plugin sync, and CLAUDE.md auto-discovery, so it is only safe for the standalone one-shot test script, NEVER for a real interactive Codeman session (which depends on hooks for idle detection, trust-dialog auto-accept, etc. — see the External CLI modes section of CLAUDE.md). Whether an INTERACTIVE claude session with a custom model hits the same hang (vs. just a background warning) is untested and should be checked before calling chunk 5/6 done for claude |
|
||||
| `opencode` | `OPENCODE_CONFIG_CONTENT` env var (already a registry mechanism, `stock.ts:342`) holding a JSON blob: `{"provider":{"custom":{"options":{"baseURL":...,"apiKey":...},"models":{"<name>":{}}}},"model":"custom/<name>"}` | **Verified by user** |
|
||||
| `codex` | TOML `config.toml`: top-level `model = "<id>"` + `[model_providers.custom]` (`base_url`, `env_key` naming an env var the real API key rides in — never a literal TOML field, since codex's schema has no such field). Written to an isolated dir via `CODEX_HOME` (`stock.ts:405-415`) so the user's own `~/.codex/config.toml` is never touched | **Config STRUCTURE verified** against a real codex binary (an earlier `[model].default` table shape was rejected: "invalid type: map, expected a string" — caught live). **Protocol picture more nuanced than a flat break, re-verified live twice on 2026-09-17 against a llama-swap deployment that DOES answer `/v1/responses`** (an earlier test's `Reconnecting...`/`high demand` failure does not reproduce against every llama-swap setup): a plain, no-tool-call chat turn (`codex exec 'reply with just OK'`) returned a real reply. But a real tool-call attempt (`run the shell command: echo hello`) came back as an `agent_message` TEXT item — the tool-call JSON printed as the model's answer, not a `function_call` item codex would actually execute (confirmed via `codex exec --json`'s raw event stream: `item.completed`/`agent_message`, never `function_call`). Since tool execution is what makes codex a coding agent at all, this remains **not usable for real work**, just with a different, more specific failure mode than previously documented — still do not present this as working. Separately, EVERY custom-endpoint codex session also prints `warning: Model metadata for '<id>' not found. Defaulting to fallback metadata...` on launch (confirmed harmless — the successful plain-text reply above still had it): codex's per-model metadata (reasoning tiers, system-prompt templates, context-window figures) comes from `models_cache.json`, a LOCAL CACHE of OpenAI's own hosted model catalog that a custom model can never appear in by construction. No config.toml override exists for it, and the isolated `CODEX_HOME` never gets a `models_cache.json` written into it at all (confirmed: inspected a live, actively-used isolated dir — codex evidently can't reach OpenAI's catalog endpoint for this session and just falls back silently every time, with no file left behind to fix or clean up). Fabricating a fake catalog entry to suppress the warning would mean copying the _shape_ of OpenAI's own proprietary schema — including their real per-model system-prompt content, visible in a genuine `models_cache.json` — for a warning confirmed to have no effect on the actual (broken) tool-calling outcome; not worth building |
|
||||
| `gemini` | Env vars `GOOGLE_GEMINI_BASE_URL` + `GEMINI_API_KEY` + `GEMINI_MODEL`; CLI needs a restart to pick them up | **Confirmed BROKEN against llama.cpp/llama-swap, unresolved after real investigation.** Setting `GOOGLE_GEMINI_BASE_URL` makes gemini-cli internally select an `AuthType.GATEWAY` auth path (undocumented — inferred from behaviour) with validation requirements distinct from every normal auth mode; a real run against llama-swap fails with `Invalid auth method selected` regardless of what key/format is supplied. Tried and all failed: a Google-format dummy API key, `GOOGLE_GENAI_USE_VERTEXAI=false`, a `GEMINI_DEFAULT_AUTH_TYPE` override, and hand-writing `settings.json` directly. `--skip-trust` was a real, separate fix (without it a trust-folder check silently overrides `--approval-mode yolo` back to `default`) but does not touch this auth failure. Documented as an open gap, not shipped as working — the registry entry and injection code exist and are exercised by the test script, but end-to-end gemini support needs upstream investigation of `GATEWAY` AuthType before it can be called done |
|
||||
| `pi` | Config file `~/.pi/agent/models.json` with a custom provider whose `models` is an **array** of `{id}` objects (not an object keyed by id) plus `authHeader: true`. Redirected via the child process's own `HOME` env var, isolated per test/session — **not** `PI_CONFIG_DIR`, which does nothing for pi (grepped pi's entire bundled JS source: the string appears nowhere) | **Verified end-to-end** against a real llama-swap server — real "hello world" reply came back. Two real bugs found and fixed before this worked: (1) `PI_CONFIG_DIR` is not read by pi at all — pi hardcodes `~/.pi/agent/models.json` with no dedicated override, so the actual redirect has to be the child process's `HOME`; (2) `models` must be an array of `{id}` objects per pi's own bundled `docs/models.md`, not an object keyed by model id (silently loaded zero models). Also requires an explicit `--model custom/<id>` on invocation — without it pi falls back to its own default provider and fails with "No API key found for the selected model" |
|
||||
| `grok` | TOML `config.toml`: a fixed `[model.codeman-custom]` block (`base_url`, `env_key` naming an env var the key rides in, never a literal TOML field) written to an isolated dir via `GROK_HOME`. Invoked with `-m codeman-custom` | **Verified end-to-end** against a real llama-swap server — real "hello world" reply came back. The ORIGINAL recipe in this table (env vars `GROK_BASE_URL`/`XAI_API_KEY`/`GROK_MODEL`) was flat-out **wrong**, not just unverified: it produced "Not signed in" against a real binary. Grok's real mechanism, confirmed against xAI's own docs and a live binary, is a `config.toml` with a `[model.<name>]` block, redirected via `GROK_HOME`; the key still rides as an env var (`XAI_API_KEY` via `env_key`), just referenced from the TOML rather than read directly |
|
||||
| `deepseek` | Reuse the **existing** `DEEPSEEK_BASE_URL` + `DEEPSEEK_API_KEY` keys (already declared in `stock.ts`), now with `appendV1Suffix: true` (see confidence). Only `DEEPSEEK_BASE_URL` is in `privilegedEnvKeys` — `DEEPSEEK_API_KEY` deliberately stays clamp-exempt, since a non-granted owner supplying their OWN key removes privilege rather than granting it (adding it to the clamp list was a real regression, caught by `test/deepseek-mode.test.ts` and fixed before merge). No model-selection var — dsh model is a profile composition entry, not a flag/env var | **Root cause of the original `HTTP_404` found and fixed, by reading dsh's own bundled source — the same bar pi/grok's fixes were held to.** Installed `@deepseek-ai/dsh` (all its real published dependencies) into a scratch directory purely to read `@deepseek-ai/dsh-llm-deepseek/lib/index.js`: it builds its request as `fetch(\`${connection.baseURL}/chat/completions\`, ...)`with`baseURL`read straight from`DEEPSEEK_BASE_URL`(or defaulting to DeepSeek's real public API root,`https://api.deepseek.com`, which also carries no `/v1`) — no `/v1` insertion of dsh's own, unlike the OpenAI-SDK convention this recipe originally assumed. llama-swap/llama.cpp only ever serves the OpenAI-conventional `/v1/chat/completions`. Confirmed live: `POST <baseUrl>/chat/completions` → `404`, `POST <baseUrl>/v1/chat/completions` → `200`, on the exact same endpoint — and dsh's own error-message template, `DeepSeek API error (HTTP ${status})`, reproduces the originally reported `dsh: HTTP_404: DeepSeek API error (HTTP 404)` precisely. Fixed by adding `appendV1Suffix` (env kind only, deepseek's entry alone — claude/gemini must NOT get it, since claude was already confirmed working against the unmodified `baseUrl`), which runs `endpoint.baseUrl` through the same `withV1Suffix()` helper `configDir`-kind CLIs already use. ⚠️ Not yet re-run end-to-end with a real `dsh` binary — no install available in this environment (no npm-installed CLI binary in `PATH`, and the `codeman-test-picker` container doesn't bundle it either); the fix is source-confirmed and live-verified at the HTTP level, but a genuine "hello world" reply through `dsh` itself is the remaining step before promoting this to **verified** alongside claude/opencode/pi/grok/omp |
|
||||
| `omp` | Config file `~/.omp/agent/models.yml` with the same array-shaped `models` + `authHeader: true` fix as pi. Redirected via `HOME`, same reasoning as pi (`PI_CONFIG_DIR` does not relocate omp's config either, despite an earlier CLAUDE.md note claiming it does) | **Verified end-to-end** against a real llama-swap server — real "hello world" reply came back, after applying the same two fixes as pi (array-shaped `models`, `HOME`-redirect instead of `PI_CONFIG_DIR`) plus an explicit `--model custom/<id>` on invocation. Unverified against omp's own official docs (none are bundled in the install), but empirically confirmed working live |
|
||||
| `antigravity` | No CLI/env/config mechanism found — Antigravity's docs describe only a GUI settings panel, and explicitly say a custom endpoint "cannot currently" become the core reasoning model. **Not implemented**; toolbar entry stays disabled for this mode with an explanatory tooltip | No known mechanism |
|
||||
|
||||
Everything web-researched-but-unverified gets implemented but must be
|
||||
smoke-tested against real installs of those CLIs before being called done —
|
||||
@@ -208,6 +208,14 @@ extra per-model configuration on Codeman's side at all.
|
||||
|
||||
### 4. Toolbar UI
|
||||
|
||||
> **Superseded.** This section describes the toolbar-button design as originally
|
||||
> planned. What actually shipped is a Run-menu picker instead: one generated entry
|
||||
> per (capable harness, saved endpoint) pair directly in the existing `#runModeMenu`
|
||||
> dropdown, rather than a separate `#customModelBtn`/`#customModelMenu` surface. See
|
||||
> [`docs/custom-model-endpoints.md`](custom-model-endpoints.md#the-run-menu-picker)
|
||||
> for the current design; the sections below (session-restart mechanics, security)
|
||||
> remain accurate regardless of which UI calls the underlying route.
|
||||
|
||||
- New header/toolbar button (e.g. `#customModelBtn`, `btn-toolbar
|
||||
btn-custom-model`), marker-hidden by default (`btn-custom-model--hidden`)
|
||||
and revealed by `applyHeaderVisibilitySettings()` only when
|
||||
@@ -344,8 +352,12 @@ pure unit tests and the live manual checks in Verification:
|
||||
up automatically with zero edits to the script). Already run to
|
||||
completion against the author's llama-swap server (a LAN address,
|
||||
inside a `codeman/agent:llm-test` Docker image with all 9 CLI binaries):
|
||||
claude/opencode/pi/grok/omp **PASS**, codex **FAILs as expected**
|
||||
(Responses-API protocol gap, not a bug), gemini/deepseek **UNCONFIRMED**
|
||||
claude/opencode/pi/grok/omp **PASS**, codex **partially works and still
|
||||
isn't usable** (plain chat succeeds against a llama-swap deployment that
|
||||
answers `/v1/responses`, but a real tool-call attempt comes back as
|
||||
inert text rather than an executable `function_call` — see the
|
||||
confidence table row for the full, re-verified picture), gemini/deepseek
|
||||
**UNCONFIRMED**
|
||||
(reach the server, fail for undiagnosed reasons — see their table rows),
|
||||
antigravity **SKIP** (no mechanism). Re-run this against a real cloud
|
||||
endpoint (e.g. an Azure AI Foundry deployment) once one is available, to
|
||||
|
||||
+406
-18
@@ -11,20 +11,22 @@ company gateway) — anything answering `GET /v1/models` and
|
||||
recipe confidence table, and security reasoning:
|
||||
[`custom-model-endpoints-plan.md`](custom-model-endpoints-plan.md).
|
||||
|
||||
> **Status**: backend is implemented and tested (registry capability, the
|
||||
> injection engine, the endpoint store + discovery route, the session
|
||||
> restart route). The toolbar picker / settings UI described below as the
|
||||
> intended surface is **not yet built** — until it lands, use the HTTP API
|
||||
> directly (examples below). Antigravity has no known custom-endpoint
|
||||
> mechanism and is not supported.
|
||||
> **Status**: fully wired end to end — registry capability, the injection
|
||||
> engine, the endpoint store + discovery route, both the restart-in-place
|
||||
> apply route (Claude) and the one-shot quick-start launch path (every
|
||||
> other supported harness), a settings-panel CRUD surface, and the Run-menu
|
||||
> picker described below. Antigravity has no known custom-endpoint
|
||||
> mechanism and is not supported. The HTTP API (examples below) still works
|
||||
> directly and is what the picker itself calls under the hood.
|
||||
|
||||
## Turning it on
|
||||
|
||||
App Settings → Agents & CLIs → **Custom Model Endpoints** (synced setting
|
||||
`customModelEndpointsEnabled`, default **OFF**). Until the toolbar picker
|
||||
lands, nothing reads this setting: the HTTP routes below work whether it is
|
||||
on or off, and it exists now only so the picker has a switch to hang off
|
||||
when it ships. The API equivalent:
|
||||
App Settings → Models → **Custom model endpoints** (synced setting
|
||||
`customModelEndpointsEnabled`, default **OFF**). Turning it on does two
|
||||
things: it reveals the endpoint list/add/edit/discover panel in that same
|
||||
settings section, and it makes the Run menu offer a generated entry per
|
||||
(harness, endpoint) pair — see "The Run-menu picker" below. The API
|
||||
equivalent:
|
||||
|
||||
```bash
|
||||
curl -sk -X PUT https://localhost:3000/api/settings \
|
||||
@@ -34,6 +36,9 @@ curl -sk -X PUT https://localhost:3000/api/settings \
|
||||
|
||||
## Adding an endpoint
|
||||
|
||||
Via App Settings → Models → Custom model endpoints → **+ Add endpoint**, or
|
||||
directly:
|
||||
|
||||
```bash
|
||||
curl -sk -X POST https://localhost:3000/api/model-endpoints \
|
||||
-H 'Content-Type: application/json' \
|
||||
@@ -62,7 +67,173 @@ configured, `PUT`/`DELETE /api/model-endpoints/:id` update or remove one.
|
||||
Endpoint management is admin-only in multi-user mode, same as remote/docker
|
||||
hosts — these are machine-level infra, not per-user settings.
|
||||
|
||||
## Applying a model to a session
|
||||
**Context length is discovered too, opportunistically and safely.** The plain
|
||||
`GET /v1/models` response has no context-window field. Discovery only ever
|
||||
looks for one for a model llama-swap's own response already reports
|
||||
`status.value === "loaded"` for — never for an unloaded one, because
|
||||
llama-swap treats `?model=` as a routing hint and asking about a model that
|
||||
isn't loaded risks triggering an actual (slow, GPU-swapping) load as a side
|
||||
effect of what should be read-only discovery. A server with no `status` field
|
||||
on any entry at all (not llama-swap) gets no context-length enrichment,
|
||||
rather than guessing. A model's previously-learned context length survives a
|
||||
later cycle where it wasn't the loaded one; it's dropped only once the model
|
||||
disappears from the endpoint's list entirely. Stored per model in
|
||||
`modelContextLengths` and applied automatically (see "Applying a model to a
|
||||
session" below) so a CLI that would otherwise assume a large default context
|
||||
window for an unrecognized model id stops silently overflowing a much
|
||||
smaller real one.
|
||||
|
||||
**Where that number actually comes from matters, and got this wrong once
|
||||
already.** The first cut read it from llama.cpp's own
|
||||
`GET /props?model=<id>` (`n_ctx`) — plausible, and it worked in testing, but
|
||||
confirmed live to be actively WRONG for a `--fit-ctx`-launched llama-swap
|
||||
backend: `/props` reported `n_ctx: 154112` for a model llama-swap itself had
|
||||
launched with `--fit-ctx 16384`, and the real server then refused a request
|
||||
right at that real 16384-token limit — `/props`'s `n_ctx` appears to report
|
||||
the model's theoretical/trained maximum there, not the runtime-configured
|
||||
one. Discovery now parses the REAL configured size straight out of
|
||||
llama-swap's own launch command instead (`GET /running`'s `cmd` field —
|
||||
`--fit-ctx <N>` first, then the plain llama.cpp `-c`/`--ctx-size` a
|
||||
hand-written command might use), and only falls back to the `/props` probe
|
||||
when `cmd` states no recognizable flag at all.
|
||||
|
||||
**File size is discovered too, when the server states one.** llama-swap
|
||||
writes a GB figure into an auto-discovered model's own `description`
|
||||
(`"Auto-discovered 16.35 GB - parameters auto-fitted by llama.cpp"`), parsed
|
||||
into `modelSizesGB` — unlike context length, this needs no `/props` probe
|
||||
(the figure is right there in the `/v1/models` response) and so is populated
|
||||
for every model regardless of loaded state. A hand-configured profile's own
|
||||
description has no such figure and correctly gets no entry, never a guess.
|
||||
Used only to label the Run-menu picker's "loading model" banner (e.g.
|
||||
"Loading qwen3.8-27b-ud-q4_k_xl (16.4 GB) on llama-swap..."); never anything
|
||||
a server-side check relies on.
|
||||
|
||||
**The loading banner is unbounded by design, and says so — no countdown, no
|
||||
automatic give-up.** An earlier version scaled an expected-time estimate and
|
||||
a timeout off the model's file size and auto-closed the session once that
|
||||
elapsed, but a real load's actual duration depends on hardware this feature
|
||||
has no way to know (VRAM, storage speed, whatever else is contending for the
|
||||
GPU) — any fixed number was a guess dressed up as a fact, and a model that
|
||||
genuinely takes 10+ minutes on slower hardware would just get killed
|
||||
mid-load by its own display. The banner now says outright that it can take a
|
||||
while depending on hardware and model size, polls
|
||||
`GET /api/model-endpoints/:id/running-status` every second for as long as it
|
||||
takes, and carries a **Cancel** button (rendered on the banner itself) that
|
||||
ends the wait and closes the session the load was for — the user's own call
|
||||
on when it's taking too long, not a fixed number baked into the client.
|
||||
|
||||
**The banner's second line is the real backend log line, not a guess.**
|
||||
llama-swap's `GET /api/events` SSE stream carries the actual `llama-server`
|
||||
process's own stdout — `load_model: loading model '<path>'`,
|
||||
`llama_server: model loaded`, tokenizer warnings, all of it — tagged
|
||||
`source: "upstream"`, distinct from llama-swap's own `source: "proxy"`
|
||||
request-access lines. `running-status`'s response now includes `logLine`
|
||||
(via `getLatestLlamaSwapLogLine`), and the banner shows it on its own line
|
||||
under the disclaimer, e.g. "llama.cpp: load_model: loading model '...'" —
|
||||
confirmed live end-to-end through a real forced swap, sequentially showing
|
||||
the model path, a tokenizer warning, then staying on whatever llama.cpp last
|
||||
printed once the load goes quiet (never cleared back to blank). ⚠️
|
||||
**`GET /logs` — the endpoint this feature's own first cut was built
|
||||
against — turns out to carry ONLY llama-swap's own proxy request-access
|
||||
log.** Confirmed live it never showed a single backend line, even seconds
|
||||
after a real, verified model swap; `/api/events`'s `logData` frames are the
|
||||
only source that actually has it, and its own `source` field (`upstream` vs
|
||||
`proxy`) is what `getLatestLlamaSwapLogLine` filters on. One `/api/events`
|
||||
connection is held open per endpoint and reused across every session
|
||||
watching a load on it (confirmed live to stay open indefinitely, unlike
|
||||
`/logs`, which closes after a fixed ~100KB), idle-closed after 30s of nobody
|
||||
polling it (`pruneIdleLlamaSwapLogTails`, same 20s sweep as the
|
||||
swap-displacement check below).
|
||||
|
||||
`defaultModelId` names which discovered model the picker pre-marks for that
|
||||
endpoint — the settings panel's Edit form exposes it as a select populated
|
||||
from the endpoint's own discovered `models`, and the route refuses a value
|
||||
that isn't one of them. It is applied automatically only when the endpoint
|
||||
has exactly one discovered model (nothing to choose); with two or more it
|
||||
is a pre-selection in the model-picker dialog below, never a silent default.
|
||||
Re-discovering drops a default that no longer appears in the fresh list
|
||||
rather than carrying an invalid one forward.
|
||||
|
||||
**Model lists refresh themselves.** A background sweep (`server.ts`,
|
||||
`CUSTOM_MODEL_REDISCOVER_INTERVAL_MS`, every 5 minutes) re-discovers every
|
||||
saved endpoint the same way the manual `POST .../discover-models` route
|
||||
does, best-effort per endpoint — one being unreachable on a given cycle
|
||||
never blocks the others. Off under `npm test`, same reasoning as the Codex
|
||||
plan-usage poll it sits beside: no real network to hit, no server instance
|
||||
to keep the timer alive for.
|
||||
|
||||
## The Run-menu picker
|
||||
|
||||
With the setting on and at least one endpoint carrying a discovered model,
|
||||
the toolbar's Run dropdown grows a **Custom Endpoints** section: one entry
|
||||
per (harness that can redirect to a custom endpoint, saved endpoint) pair,
|
||||
e.g. "Claude Code (llama.cpp)". The harness list is read off the CLI
|
||||
registry's own `capabilities.customModelInjection` at page render
|
||||
(`window.__codemanCustomModelClis`, `server.ts`) — never a hardcoded id list
|
||||
in the frontend — so a CLI whose injection recipe lands later shows up with
|
||||
no frontend change, and Antigravity (`unsupported`) never does.
|
||||
|
||||
Picking an entry re-fetches the endpoint (`selectCustomModelEntry()`,
|
||||
`session-ui.js`) rather than trusting anything cached from the dropdown's
|
||||
own render — the model list can have changed via the 5-minute sweep above
|
||||
or a settings-panel edit since the menu opened. With exactly one discovered
|
||||
model it runs straight away; with two or more, a small modal
|
||||
(`#customModelPickModal`) lists them and asks which one to use for this
|
||||
launch, with the endpoint's `defaultModelId` marked but not auto-chosen —
|
||||
the point of asking is letting one launch deliberately differ from the
|
||||
saved default, not just confirming it.
|
||||
|
||||
**How the launch itself applies the endpoint depends on the harness.** For
|
||||
opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP (`runCustomModelEntry` →
|
||||
`_runCustomModelEntryOneShot`), the endpoint/model is folded into the SAME
|
||||
`POST /api/quick-start` call that creates the session (`customModel` field),
|
||||
so the session launches directly on the endpoint — no restart, no visible
|
||||
relaunch. Claude (`_runCustomModelEntryViaRestart`) still uses the original
|
||||
two-step design: the launch runs a single native session exactly the way its
|
||||
own Run-menu entry would, then **waits for the new session to go idle**
|
||||
(`GET .../wait?until=idle`, bounded at 20s — a normal 200 either way, never
|
||||
an error, per the wait endpoint's own contract) before applying the endpoint
|
||||
via the restart route below. That wait exists because a freshly launched CLI
|
||||
reports itself as `busy` for its own startup (a boot spinner, a
|
||||
workspace-trust check) well before the apply call would otherwise reach it,
|
||||
and the apply route correctly refuses to restart a session mid-turn — a
|
||||
fresh boot looks exactly like one from the outside. A session still busy
|
||||
after the wait reaches the apply call anyway and gets that route's own
|
||||
honest `SESSION_BUSY` error, now visible as a sticky toast with a close
|
||||
button rather than a generic message that vanished in three seconds. Claude
|
||||
stays on this path because its own restart (`--resume`-based, keeping the
|
||||
conversation) is far less jarring than the other seven's, and `runClaude()`'s
|
||||
multi-tab launch and docker-config-drift confirm/retry loop make folding it
|
||||
into the one-shot path separate work. It is a
|
||||
one-off "try this endpoint" action, not a sticky mode: the plain Run button
|
||||
still means "this harness, native cloud" afterward. Entries are hidden
|
||||
entirely for a remote or Docker active case, since the apply route refuses
|
||||
both (see the next section).
|
||||
|
||||
## Launching directly on an endpoint (no restart)
|
||||
|
||||
```bash
|
||||
curl -sk -X POST https://localhost:3000/api/quick-start \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"caseName": "myapp", "mode": "codex", "customModel": {"endpointId": "llama-box", "modelId": "qwen3"}}'
|
||||
```
|
||||
|
||||
`POST /api/quick-start`'s `customModel` field (`{endpointId, modelId,
|
||||
confirmed?}`) computes the same injection the restart route below does, but
|
||||
BEFORE the session exists — the session is minted its own id up front
|
||||
(`crypto.randomUUID()`), the injection (env vars, and for a `configDir`-kind
|
||||
CLI, the written config file) targets that real id, and the session launches
|
||||
already pointed at the endpoint. No restart, because there was never a
|
||||
native-backend launch to restart away from. Runs the same llama-swap
|
||||
conflict check as the restart route (below) — a `409`-shaped
|
||||
`{requiresConfirmation, currentlyLoadedModel, affectedSessions}` response
|
||||
with no session created, resolved by retrying with `confirmedSwap: true` — and
|
||||
is refused the same way for a remote or Docker case. This is what the
|
||||
Run-menu picker uses for opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP;
|
||||
Claude still uses the restart route below (see "The Run-menu picker" above
|
||||
for why).
|
||||
|
||||
## Applying a model to an ALREADY-RUNNING session
|
||||
|
||||
```bash
|
||||
curl -sk -X POST https://localhost:3000/api/sessions/<sessionId>/custom-model \
|
||||
@@ -84,6 +255,179 @@ since for those three the config file alone does not switch the model.
|
||||
reattaches the durable remote/in-container tmux rather than relaunching the
|
||||
agent, so the selection would report success and change nothing.
|
||||
|
||||
**Claude gets two more env vars when known/applicable, both declared on its
|
||||
registry entry (`contextLengthVar`/`configDirVar`), not hardcoded here:**
|
||||
|
||||
- `CLAUDE_CODE_MAX_CONTEXT_TOKENS` is set to `modelId`'s discovered context
|
||||
length (see the discovery section above) whenever one is known. Without
|
||||
it, Claude Code assumes a large (200k) window for any unrecognized custom
|
||||
model id and never compacts, which reliably overflows a much smaller real
|
||||
local context — confirmed live: a stock ~33.7K-token system prompt against
|
||||
a 16384-token llama-swap model failed with `exceeds the available context
|
||||
size`. No entry for the model in `modelContextLengths` means the var is
|
||||
simply omitted, never a guess. ⚠️ **This var only affects when Claude
|
||||
Code compacts conversation _history_ — it cannot fix a model whose real
|
||||
context is smaller than Claude Code's own fixed per-turn overhead**
|
||||
(system prompt + tool schemas, empirically ~36.4K tokens, confirmed live
|
||||
via an `in:0 out:0` failure on the very first message, before any
|
||||
history exists to compact). No context-length declaration changes that
|
||||
fixed overhead, so a model below the safe floor fails outright on
|
||||
message one regardless of what this var says. See "Context-window floor
|
||||
warning" below for how Codeman catches this case before launching
|
||||
instead of after.
|
||||
- `CLAUDE_CONFIG_DIR` is pointed at the same isolated per-session directory
|
||||
the `configDir`-kind CLIs use (empty, no files written into it), so the
|
||||
injected `ANTHROPIC_API_KEY` never shares a directory with a stored
|
||||
claude.ai OAuth login. Claude Code still prints "Both claude.ai and
|
||||
ANTHROPIC_API_KEY set" when the two coexist in the same config directory —
|
||||
cosmetic (confirmed live: the API key wins for actual requests either way,
|
||||
visible in the terminal's own `API Usage Billing` line) but worth
|
||||
eliminating rather than living with. The directory's `projects`
|
||||
subdirectory is symlinked (a junction on Windows) back to the real
|
||||
`~/.claude/projects` so the response viewer, subagent windows and Read My
|
||||
Mind keep working for that session — the same trade-off and fix documented
|
||||
for a manually-set `CLAUDE_CONFIG_DIR` in
|
||||
[`docs/wiki/Agent-CLIs.md`](wiki/Agent-CLIs.md), just applied
|
||||
automatically here. Best-effort: a platform that refuses the symlink keeps
|
||||
the pre-existing blind-response-viewer side effect rather than failing the
|
||||
whole custom-model apply over it. ⚠️ **This relocates the whole `.claude`
|
||||
tree, not just transcripts**: a custom-model Claude session also loses the
|
||||
user's global `settings.json`, user-level skills (the codeman agent skill
|
||||
included), user-level agents and commands, and the MCP servers configured
|
||||
in `~/.claude.json` — none of those are symlinked back, only `projects` is.
|
||||
A fine trade for "point this session at my local llama.cpp," but worth
|
||||
knowing before it surprises you mid-session.
|
||||
|
||||
**That isolated directory needed one more fix to actually be usable
|
||||
non-interactively.** An otherwise-empty `CLAUDE_CONFIG_DIR` has none of a
|
||||
real profile's prior "Detected a custom API key — use it?" approvals, so
|
||||
without more, Claude Code stops and asks that on _every single launch_ —
|
||||
confirmed live, and with nobody at a TTY to answer, its own default answer
|
||||
("No") silently refuses the very key this feature just injected, which
|
||||
looks like the endpoint being ignored entirely. `customModelInjection`'s
|
||||
`apiKeyTrustFile` (`{ relPath: '.claude.json', shape:
|
||||
'claude-api-key-responses' }` on claude's entry) pre-seeds that exact
|
||||
approval: the apply step merges `customApiKeyResponses.approved: [apiKey]`
|
||||
into `<configDir>/.claude.json`, the same field a real answered prompt
|
||||
itself writes to (confirmed against a real file after answering by hand
|
||||
once) — this answers the prompt in advance rather than bypassing it. The
|
||||
merge preserves whatever else the CLI already wrote into that file on an
|
||||
earlier launch in the same isolated directory (`userID`, `numStartups`,
|
||||
earlier approved keys), and a missing or corrupt file is treated as empty
|
||||
rather than failing the apply.
|
||||
|
||||
**A fresh `CLAUDE_CONFIG_DIR` isn't just missing that one approval — Claude
|
||||
Code treats it as a brand-new profile and replays its ENTIRE first-run
|
||||
sequence on every launch: the theme picker, the security-notes screen, the
|
||||
per-project "trust this folder?" dialog, and (running with
|
||||
`--dangerously-skip-permissions`) a one-time warning about bypassing
|
||||
permissions.** Confirmed live: none of these show up again for a real,
|
||||
already-onboarded profile, but every custom-model session gets a fresh,
|
||||
otherwise-empty isolated directory, so it saw all four every single time.
|
||||
`customModelInjection`'s `skipFirstRunPrompts` (`true` on claude's entry,
|
||||
requires `apiKeyTrustFile` since it reuses the same file) pre-seeds the
|
||||
state a real profile accumulates from answering all of that once:
|
||||
`hasCompletedOnboarding: true` and the launching session's own
|
||||
`projects[workingDir].hasTrustDialogAccepted: true` go into the same
|
||||
`<configDir>/.claude.json` the API-key approval above already merges into
|
||||
(other projects, and other fields on this session's own project entry, are
|
||||
left untouched), and `skipDangerousModePermissionPrompt: true` goes into
|
||||
`<configDir>/settings.json` — a different file, merged the same
|
||||
corrupt-tolerant way. `workingDir` is used exactly as the session was
|
||||
launched with as its cwd, never realpath'd or slash-normalized, since
|
||||
that's the literal string Claude Code itself uses as the project key.
|
||||
|
||||
**llama-swap gets two more fixes on top of the context-length/config-dir
|
||||
ones above, both from watching a real switch live.** llama.cpp only ever
|
||||
runs one model at a time; llama-swap swaps the backing process on demand,
|
||||
which can take anywhere from a few seconds to well over a minute:
|
||||
|
||||
- **The conflict check.** Both apply routes (the restart one here and the
|
||||
one-shot `POST /api/quick-start` above) call llama-swap's own
|
||||
`GET /running` first — feature-detected, so a plain llama.cpp/OpenAI-
|
||||
compatible server (no such endpoint) is simply never checked. If a
|
||||
_different_ model is currently loaded and ready, and another **live
|
||||
session's own selection** is using it, the apply returns
|
||||
`{requiresConfirmation: true, currentlyLoadedModel, affectedSessions}`
|
||||
instead of silently switching — nothing is applied or created yet.
|
||||
Retrying with `confirmedSwap: true` skips the check (the legacy `confirmed: true`
|
||||
still means both questions). Switching with nothing
|
||||
else affected proceeds immediately; this is a warning about disrupting
|
||||
another session, never a gate on the switch itself.
|
||||
- **Actually starting the load.** llama-swap has no "switch model" admin
|
||||
call — the only thing that starts a swap is a real inference request
|
||||
naming the model, and confirmed live: applying a selection alone never
|
||||
reached llama-swap at all (nothing in its own server logs), since nothing
|
||||
had actually asked it to load anything yet. Both apply routes now also
|
||||
send the smallest real request that will —
|
||||
`POST <baseUrl>/v1/chat/completions` with `max_tokens: 1` and one
|
||||
throwaway message — whenever the
|
||||
target model isn't already the one loaded and ready, fire-and-forget (its
|
||||
response is never read; `GET /api/model-endpoints/:id/running-status`,
|
||||
polled client-side, is what actually confirms readiness). The response
|
||||
also carries `modelSwapInProgress: true` in that case, which is what
|
||||
drives the Run-menu picker's own "loading model" status banner.
|
||||
|
||||
## Catching a swap after the fact
|
||||
|
||||
The conflict check above only runs at the moment a session is created or a
|
||||
model is applied — it has no way to catch a swap that happens **later**.
|
||||
Confirmed live: a session created while nothing else conflicted at that
|
||||
exact instant can still get silently displaced afterward, once a
|
||||
_different_ session's own normal use (or its own create-time load trigger)
|
||||
asks llama-swap to load something else. llama-swap has no push
|
||||
notification of its own for this, so a background sweep
|
||||
(`detectCustomModelSwapDisplacements`, `CUSTOM_MODEL_SWAP_CHECK_INTERVAL_MS`
|
||||
= 20s in `server.ts`) polls `GET /running` once per distinct endpoint that
|
||||
has at least one live custom-model session, and compares each such
|
||||
session's own `modelId` against what is actually loaded. A session whose
|
||||
model is no longer in that list gets a `custom-model:swapped-out` SSE event
|
||||
(`{sessionId, sessionName, endpointId, previousModel, currentlyLoadedModel}`),
|
||||
shown as a global toast — global rather than tied to that session's tab,
|
||||
since the whole point is telling the user before they type into it
|
||||
expecting the model they picked. Notifies **once per displacement**: the
|
||||
same de-dupe `Set` clears a session's flag once its own model is loaded and
|
||||
ready again, so a later, genuinely new displacement notifies again rather
|
||||
than the session staying silently un-notified forever after the first one.
|
||||
|
||||
## Context-window floor warning
|
||||
|
||||
Claude Code's own fixed per-turn overhead (system prompt + tool schemas,
|
||||
empirically ~36.4K tokens) can exceed a small local model's _entire_ real
|
||||
context on its own, before any conversation history exists to fill it —
|
||||
confirmed live twice, both as an `in:0 out:0` failure on the very first
|
||||
message sent. `CLAUDE_CODE_MAX_CONTEXT_TOKENS` (above) cannot fix this: it
|
||||
only governs when Claude Code compacts conversation history, and there is
|
||||
no history yet on message one. Applying such a model would look like the
|
||||
endpoint being ignored, or the wrong model being used, when in fact the
|
||||
endpoint applied correctly and the model is simply too small for this CLI.
|
||||
|
||||
Both apply routes (the restart route and the one-shot `POST
|
||||
/api/quick-start`) now check for this **before** launching or restarting
|
||||
anything, gated on the CLI's registry entry declaring a `contextLengthVar`
|
||||
(currently only claude — the check is a no-op for every other CLI by
|
||||
construction, never a hardcoded mode check). If the model's discovered
|
||||
context (`modelContextLengths`, from discovery above) is below
|
||||
`CLAUDE_MIN_SAFE_CONTEXT_TOKENS` (40000, comfortably above the measured
|
||||
~36.4K overhead), the response is `{requiresContextWarning: true, modelId,
|
||||
contextLength, minSafeContextTokens}` instead of applying — nothing is
|
||||
restarted or created yet. A context length that was never discovered at
|
||||
all skips the check entirely (nothing to compare, so it fails open rather
|
||||
than warning on every model an endpoint hasn't reported a size for).
|
||||
Retrying with `confirmedContext: true` launches anyway (the legacy `confirmed: true` still means both questions).
|
||||
|
||||
The Run-menu picker shows this as an in-app modal
|
||||
(`#customModelContextWarningModal`, matching the llama-swap conflict
|
||||
modal's look) naming the model, its discovered context, and the safe
|
||||
floor, and explaining the fix: reconfigure llama-swap to give that model
|
||||
(or a smaller one) an explicit larger context instead of relying on
|
||||
auto-fit (`--fit-ctx`), which optimizes for the biggest _model_ that fits
|
||||
rather than the biggest _context_ — e.g. adding `-c 65536` (or as large a
|
||||
`--ctx-size` as the hardware holds) to that model's llama-swap config
|
||||
entry. A smaller model at a much larger explicit context often fits in
|
||||
the same VRAM a bigger model's auto-fit context gets shrunk to make room
|
||||
for.
|
||||
|
||||
Clear back to the harness's native cloud default with:
|
||||
|
||||
```bash
|
||||
@@ -101,6 +445,14 @@ id, model and injected key NAMES are persisted, the values are re-derived
|
||||
from the endpoint store on recovery, and the pane keeps running against the
|
||||
endpoint in between because tmux retains its environment.
|
||||
|
||||
⚠️ Clearing removes injected keys **by name**, and `CLAUDE_CONFIG_DIR` is one
|
||||
of the names claude's selection injects — so a session that ALSO had
|
||||
`CLAUDE_CONFIG_DIR` set through the generic `envOverrides` field (the
|
||||
per-client-account case) loses that override on clear too, and silently
|
||||
falls back to the server's default Claude account. If you route a session
|
||||
to a specific account this way, re-apply the override after clearing a
|
||||
custom-model selection from it.
|
||||
|
||||
**New sessions always default back to the harness's native backend.** A
|
||||
custom-endpoint selection is a per-session choice, never a sticky global
|
||||
default — starting a fresh session doesn't inherit whatever the last one was
|
||||
@@ -115,17 +467,53 @@ automatically). Results:
|
||||
|
||||
- **Claude, opencode, Pi, Grok, OMP** — verified: a real "hello world" reply
|
||||
came back through the endpoint.
|
||||
- **Codex** — the config is structurally correct, but Codex only speaks the
|
||||
Responses API since Feb 2026, which llama.cpp/llama-swap don't implement.
|
||||
This is a real protocol incompatibility, not a bug here; Codex support
|
||||
needs a Responses-API-compatible endpoint.
|
||||
- **Codex** — the config is structurally correct, and against a llama-swap
|
||||
server that DOES answer `/v1/responses` (confirmed live: a plain,
|
||||
no-tool-call chat turn returned a real reply), the picture is more
|
||||
nuanced than a flat failure. A real tool-call attempt (`run the shell
|
||||
command: echo hello`) came back as `agent_message` TEXT — literally the
|
||||
tool-call JSON printed as the model's answer — instead of a
|
||||
`function_call` item Codex would actually execute (confirmed via `codex
|
||||
exec --json`'s raw event stream). So plain chat can work while the thing
|
||||
that makes Codex a coding agent — actually running commands and editing
|
||||
files — does not; treat Codex as still unreliable for real work against a
|
||||
llama.cpp/llama-swap endpoint, tool-calling gap included, not just the
|
||||
earlier-documented `wire_api` mismatch (which not every deployment hits
|
||||
the same way — some legitimately have no `/v1/responses` route at all).
|
||||
Separately, EVERY custom-endpoint Codex session prints `Model metadata
|
||||
for '<id>' not found. Defaulting to fallback metadata...` on launch —
|
||||
confirmed harmless (the reply above still came back correctly): Codex's
|
||||
model metadata (reasoning-tier options, per-model system-prompt
|
||||
templates, context-window figures) comes from `models_cache.json`, a
|
||||
local cache of OpenAI's own hosted model catalog that a custom local
|
||||
model can never appear in by construction, since it isn't one of
|
||||
OpenAI's models. There's no config.toml override for a model's metadata,
|
||||
and fabricating a fake catalog entry would mean copying the _shape_ of
|
||||
OpenAI's own proprietary schema (their per-model system-prompt content
|
||||
included) for a warning that doesn't otherwise affect behavior — not
|
||||
something to build into discovery.
|
||||
- **Gemini** — fails with `Invalid auth method selected`, traced to an
|
||||
undocumented `GATEWAY` auth path gemini-cli selects once
|
||||
`GOOGLE_GEMINI_BASE_URL` is set. Unresolved after real investigation
|
||||
(several auth workarounds were tried and ruled out); do not rely on
|
||||
Gemini support yet.
|
||||
- **DeepSeek** — the request reaches the server (env vars are read) but
|
||||
gets a consistent `HTTP_404`. Root cause not identified; best-effort only.
|
||||
- **DeepSeek** — root cause of the `HTTP_404` found and fixed. DeepSeek
|
||||
Harness's own bundled provider module (`@deepseek-ai/dsh-llm-deepseek`)
|
||||
builds its request URL as `${DEEPSEEK_BASE_URL}/chat/completions` with no
|
||||
`/v1` insertion of its own (its real public API, `https://api.deepseek.com`,
|
||||
expects the caller's base URL to already carry any needed prefix) —
|
||||
confirmed by reading its own source and, live, that
|
||||
`POST <baseUrl>/chat/completions` 404s against llama-swap while
|
||||
`POST <baseUrl>/v1/chat/completions` succeeds; the harness's own error
|
||||
template (`DeepSeek API error (HTTP ${status})`) matches the originally
|
||||
reported symptom exactly. `customModelInjection`'s new `appendV1Suffix`
|
||||
(deepseek's entry only — claude/gemini must NOT get it, since claude was
|
||||
already confirmed working against the raw `baseUrl`) fixes it by writing
|
||||
`DEEPSEEK_BASE_URL` with `/v1` appended. Not yet re-run end-to-end with a
|
||||
real `dsh` binary (no install available in this environment) — the fix
|
||||
is source-confirmed and live-verified at the HTTP level, but a real
|
||||
"hello world" reply through `dsh` itself is still outstanding before
|
||||
calling this fully verified like the harnesses above.
|
||||
- **Antigravity** — no known custom-endpoint mechanism at all; unsupported.
|
||||
|
||||
See the confidence table in `custom-model-endpoints-plan.md` for the full detail behind
|
||||
|
||||
@@ -0,0 +1,380 @@
|
||||
# Installer v2: three questions, then a URL you can open on your phone (Plan)
|
||||
|
||||
Status: **Phase 1 IMPLEMENTED (2026-09-20)**, phases 2 and 3 open. It builds on
|
||||
`docs/tailscale-installer-plan.md` (implemented 2026-08-04), which made Tailscale a
|
||||
guided option; this round makes it the thing the install ENDS on, and makes the whole
|
||||
installer shorter to sit through. Owner decisions taken before implementation: rename
|
||||
is opt-in and **defaults to no everywhere** (the machine name is used for other things);
|
||||
the URL keeps the node name unless asked; `codeman-<hostname>` is the suggested name;
|
||||
sub-path is the default for an occupied `:443`.
|
||||
|
||||
Verification record for phase 1 (all on the maintainer's box, 2026-09-20):
|
||||
|
||||
- `test/install-sh-invariants.test.ts` (28 tests, incl. the new Tailscale safety pins)
|
||||
and the detection-parity test pass; `bash -n` passes.
|
||||
- Every new decision function driven with stubbed tailscale state under **bash 5.2 and
|
||||
bash 3.2** (the `bash:3.2` container CI uses): flags, the launch default, the serve
|
||||
shape for free / ours / occupied `:443` (all four answers plus the non-interactive
|
||||
default), the three serve commands, the rename question (Enter keeps the name; `--yes`
|
||||
and non-interactive never rename; `codeman-*` nodes are skipped; `--name` is
|
||||
sanitized), `run_step` success/failure/stdin, the unit round-trip of
|
||||
`CODEMAN_BASE_URL`/`CODEMAN_PORT`/an escaped password, and the done screen.
|
||||
- A full non-interactive install into a sandboxed `HOME` with `CODEMAN_TAILSCALE=1`:
|
||||
preflight summary, kept the existing prod mapping (no serve mutation), clone 2 s,
|
||||
`npm install` 18 s, build 23 s, symlink, done screen; `install.sh status` on a pty
|
||||
renders the QR code. Nothing on the real system changed.
|
||||
- **Sub-path mode end to end over the real tailnet**: an isolated Codeman
|
||||
(`CODEMAN_INSTANCE`, port 3999, `--base-url /codeman`) behind
|
||||
`tailscale serve --https=8445 --set-path /codeman 3999` answered `/codeman/api/status`,
|
||||
`/codeman/` (with `<base href="/codeman/">` and `__CODEMAN_BASE__="/codeman"`), the
|
||||
hashed CSS/JS, `/codeman` without a slash, and the SSE stream; mapping and server
|
||||
removed afterwards. **Correction to section 2**: serve STRIPS the mount prefix
|
||||
before proxying (a direct `/codeman/api/status` on the server is 404 while the same
|
||||
path through serve is 200). That is fine because Codeman's ingress tolerates
|
||||
unprefixed requests; `--base-url` is needed for the URLs Codeman EMITS, not for
|
||||
what it receives.
|
||||
- Not yet exercised on a fresh machine (unchanged from the previous plan): Tailscale
|
||||
absent / logged out / HTTPS toggle off, the rename against a real node (the
|
||||
off-rename-re-add order is implemented but only unit-driven), macOS, uninstall. The
|
||||
Mac mini and a throwaway VM are the venues; see section 8.
|
||||
- **Review fixes (2026-09-21)**, from the two reviews on PR #460 (DeepSeek Harness, then
|
||||
Claude): the done screen's Start line is composed from every non-default value
|
||||
(`start_command_hint`, shared with the exec branch as `export_bind_env`), so "do not
|
||||
start" under a sub-path or a custom port no longer prints a bare `codeman web`; the
|
||||
`--lan`/`--tailscale`/env preset paths keep an existing password instead of rewriting
|
||||
the unit open; `--password`/`--port` flip `RECONFIGURE` so they reach the unit;
|
||||
`install.sh name` re-syncs the unit's base URL after a rename; the sudo keepalive is
|
||||
ended before the `exec` into the foreground server; Ctrl+C in the HTTPS-toggle poll
|
||||
skips Tailscale instead of killing the run; `uninstall` asks before removing a
|
||||
LaunchDaemon it never wrote; a foreign LaunchDaemon gets a restart hint and the done
|
||||
screen stops claiming the new build is running; the preflight summary reads the
|
||||
Tailscale state without node; the LAN security notice uses the configured port; a
|
||||
bare re-run ends on the done screen; a build failure after a rename names the
|
||||
`install.sh tailscale` recovery; `TS_JOINED_HERE` is gone.
|
||||
|
||||
Goal, in one sentence: a user runs the one-liner, answers at most three questions, walks
|
||||
away during the build, and comes back to `https://<name>.<tailnet>.ts.net` printed with a
|
||||
QR code, already answering, on every device in their tailnet. That is exactly the
|
||||
maintainer's own production setup (`tnode.tailf80371.ts.net` fronting `127.0.0.1:3000`),
|
||||
and the installer should produce it without the user knowing what `tailscale serve` is.
|
||||
|
||||
## 1. Where the installer is today
|
||||
|
||||
Facts from reading `install.sh` (2886 lines, 19 `prompt_yes_no` sites) and the live
|
||||
Tailscale state on the maintainer's box (tailscale 1.102.2, user-owned node, MagicDNS +
|
||||
HTTPS certs on, serve mapping `443 -> https+insecure://localhost:3000`).
|
||||
|
||||
**The order is backwards for a human.** The flow is: detect -> ask about git -> ask about
|
||||
node -> ask about tmux -> ask about build tools -> AI CLI menu -> ask about cloudflared ->
|
||||
clone -> `npm install` -> build (minutes) -> **then** the network-access question -> the
|
||||
Tailscale sub-steps (install? login URL, sudo for operator, admin-console toggle loop) ->
|
||||
the launch menu (no default; a bare Enter re-prompts) -> tunnel-service question. A fresh
|
||||
Ubuntu server taking the Tailscale route answers roughly ten prompts plus two to four sudo
|
||||
password prompts, split around a multi-minute build. The user cannot walk away at any
|
||||
point, and the question that matters most (how do I reach it) comes last.
|
||||
|
||||
**The Tailscale flow works but was never exercised on a fresh machine.** The previous
|
||||
plan's manual matrix still lists items 1-4, 7 and 10-12 (Tailscale absent, logged out,
|
||||
HTTPS toggle off, port 443 occupied, macOS, uninstall, phone PWA) as untested. The
|
||||
maintainer's own verification was the idempotent "kept as-is" path.
|
||||
|
||||
**The URL is the machine's name, full stop.** `setup_tailscale_serve` derives it from
|
||||
`.Self.DNSName`, and nothing lets the user influence it. A second Codeman on the same
|
||||
tailnet is `macminis-mac-mini.tailf80371.ts.net`, which tells you nothing about Codeman.
|
||||
|
||||
**Port 443 taken means give up or clobber.** If another app already owns the root of
|
||||
`:443`, the only offer is "replace it?" (default no), and declining falls back to
|
||||
local-only. Codeman already supports running under a sub-path (`--base-url`), and
|
||||
Tailscale serve supports mounting a path (`--set-path`), so there is a third answer nobody
|
||||
is offered.
|
||||
|
||||
**The result is invisible afterwards.** Once the terminal scrolls away, nothing in the app
|
||||
or the CLI tells the user their Tailscale URL again. `codeman doctor` does not probe
|
||||
Tailscale; App Settings -> Remote access shows only the Cloudflare tunnel.
|
||||
|
||||
**Two service writers exist.** `install.sh` carries its own plist/unit generator (~180
|
||||
lines) next to `codeman service install` (`src/service-installer.ts`). They agree on the
|
||||
job name by design, but the bash copy is the one that writes `CODEMAN_PASSWORD` into the
|
||||
unit, so they cannot simply be merged. Left as-is in this plan (see section 9).
|
||||
|
||||
## 2. What Tailscale makes possible for the name (researched 2026-09-20)
|
||||
|
||||
| Option | Resulting URL | What it needs | Side effects | Verdict |
|
||||
| ------ | ------------- | ------------- | ------------ | ------- |
|
||||
| **A. Node name** (today) | `https://tnode.tailf80371.ts.net` | `tailscale serve --bg 3000` | none | **Default.** Zero admin-console work, matches the maintainer's prod. |
|
||||
| **B. Rename the node** | `https://codeman-tnode.tailf80371.ts.net` | `tailscale set --hostname codeman-<host>` (operator or root) | Renames the machine tailnet-wide: ssh targets, other serve URLs, the admin console entry. Tailscale de-dups a clash as `-1`. The cert follows the new name. | **Opt-in, default NO everywhere** (owner decision 2026-09-20: the machine is used for other things, so a bare Enter never renames it). The proposal was YES when the installer itself had just joined the tailnet; rejected. |
|
||||
| **C. Tailscale Service** | `https://codeman.tailf80371.ts.net` | tailscale >= 1.86 on the host; the host must have a **tag-based identity** ("You cannot use a device authenticated with a user account as a Service host"); the service is defined in the admin console first; the host is then approved there (or via `autoApprovers.services`). Public beta since 2025-10-28, all plans. | Re-authenticating a personal machine as a tagged node changes its identity (SSH ACLs, user attribution). Known daemon quirk: approval is not picked up until `serve clear` + re-advertise (tailscale/tailscale#18821). | **Detect and hint only** in this round. The maintainer's own node has `Self.Tags: null`, so it could not host one without re-tagging. Worth a real flow once someone with a tagged fleet asks. |
|
||||
| **D. Sub-path** | `https://tnode.tailf80371.ts.net/codeman` | `tailscale serve --bg --set-path /codeman 3000` plus `--base-url /codeman` on the server | Codeman runs under a prefix. Hooks are unaffected (they hit the raw port with no prefix, which `rewriteUrl` already tolerates). Serve forwards the prefix unchanged, which is exactly the shape `--base-url` was built for. | **The answer when `:443` root is already taken.** Replaces today's replace-or-nothing prompt. |
|
||||
| **E. Second port** | `https://tnode.tailf80371.ts.net:8443` | `tailscale serve --bg --https=8443 3000` | Port in the URL; the beta-preview recipe already uses this. | Fallback when the user rejects D. |
|
||||
| Funnel (public internet) | `https://tnode.tailf80371.ts.net` from anywhere | `tailscale funnel` | Public exposure; different risk class. | **Out of scope**, as before. Docs only, with the password warning. |
|
||||
|
||||
Sources: Tailscale Services docs (`tailscale.com/docs/features/tailscale-services`), the
|
||||
Services beta announcement (`tailscale.com/blog/services-beta`), machine names
|
||||
(`tailscale.com/kb/1098/machine-names`), the serve CLI reference
|
||||
(`tailscale.com/docs/reference/tailscale-cli/serve`), the macOS variants page
|
||||
(`tailscale.com/docs/concepts/macos-variants`), and `tailscale serve --help` on 1.102.2
|
||||
(which lists `--service`, `--set-path`, `--yes`, `advertise`, `get-config`/`set-config`).
|
||||
|
||||
**Trap for option B (verify on the Mac mini before shipping):** the serve config is keyed
|
||||
by `host:port` using the DNS name at configuration time (`"Web": {"tnode.tailf80371.ts.net:443": ...}`
|
||||
in `serve status --json`). Renaming a node after serve is configured most likely orphans that
|
||||
entry: the handler lookup uses the current name and never matches the old key, and the only
|
||||
tool that removes a stale key is `serve reset`, which this installer must never run. So the
|
||||
order is **rename first, then configure serve** on a fresh install, and on a retrofit
|
||||
(`install.sh name`) **turn our mapping off, rename, wait for `.Self.DNSName` to change,
|
||||
re-add**.
|
||||
|
||||
## 3. Target UX
|
||||
|
||||
### 3.1 Three questions, then walk away
|
||||
|
||||
```
|
||||
Codeman installer
|
||||
|
||||
Found: git, Node 22.14, tmux 3.4, build tools Missing: nothing
|
||||
AI CLIs: Claude Code (~/.local/bin/claude)
|
||||
Tailscale: connected as tnode (tailf80371.ts.net)
|
||||
Existing: none
|
||||
|
||||
1/3 How should the dashboard be reachable?
|
||||
1) Tailscale https://tnode.tailf80371.ts.net (recommended, already connected)
|
||||
2) Any device on your network (0.0.0.0, password required)
|
||||
3) This machine only (127.0.0.1)
|
||||
Choose [1/2/3] (default 1):
|
||||
|
||||
2/3 Name this machine "codeman-tnode" on your tailnet? [y/N]
|
||||
(only shown for option 1; default no, always)
|
||||
|
||||
3/3 Run Codeman as a background service that starts on boot? [Y/n]
|
||||
|
||||
Installing… this takes a few minutes. You can leave this running.
|
||||
✓ dependencies ✓ clone ✓ build (2m 41s) ✓ service ✓ tailscale serve
|
||||
```
|
||||
|
||||
Rules that make this work:
|
||||
|
||||
- **Every step that needs a human runs BEFORE the build.** The dependency consent, the
|
||||
AI CLI menu, the Tailscale install consent, the `tailscale up` login URL, the operator
|
||||
grant, and the tailnet HTTPS toggle all move into the question phase. The build, the
|
||||
service, `tailscale serve` and the verification are unattended.
|
||||
- **One consent for all missing system packages.** "Install git, Node 22 and build tools
|
||||
now? [Y/n]" replaces four separate prompts. Each package still runs its own
|
||||
distro-specific installer.
|
||||
- **One sudo prompt.** When anything needs root (packages, the Tailscale installer,
|
||||
`tailscale up`, the operator grant), the installer says so once, runs `sudo -v`, and keeps
|
||||
the timestamp alive in a background loop until it exits. macOS needs no sudo for the
|
||||
Tailscale GUI-app CLI and the pattern still holds for Homebrew packages.
|
||||
- **Service is the default.** Enter on the last question installs the service; "run in
|
||||
this terminal" and "don't start" stay reachable by answering, and by flag.
|
||||
- **The cloudflared question is gone from the main flow.** It is optional, defaults to
|
||||
no, and has an in-app toggle (App Settings -> Remote access). The done screen mentions it
|
||||
only when `cloudflared` is already installed. The Linux tunnel-service prompt goes with it.
|
||||
- **The HTTPS-certificates toggle no longer asks "re-check now?"** The installer prints the
|
||||
admin URL, opens it in a browser when one is available (`xdg-open` / `open`, never on a
|
||||
headless box), and polls `tailscale status --json` every 5 s for up to 5 minutes. Ctrl+C or
|
||||
the timeout falls back exactly as today.
|
||||
- **Progress, not silence.** `npm install` and `npm run build` run behind one line each
|
||||
with elapsed time; their output goes to `~/.codeman/install.log` and is printed only on
|
||||
failure, with the exact retry command.
|
||||
|
||||
### 3.2 The done screen
|
||||
|
||||
One block, the URL first, a QR code the phone can scan, and nothing the user does not need
|
||||
right now.
|
||||
|
||||
```
|
||||
✓ Codeman 1.31.0 is running
|
||||
|
||||
Your tailnet: https://codeman-tnode.tailf80371.ts.net (HTTPS, any of your devices)
|
||||
This machine: http://localhost:3000
|
||||
|
||||
▄▄▄▄▄▄▄ ▄ ▄▄ ▄▄▄▄▄▄▄
|
||||
█ ▄▄▄ █ ▄▄▀ ▄ █ ▄▄▄ █ scan with your phone
|
||||
█ ███ █ ███▀▀ █ ███ █
|
||||
█▄▄▄▄▄█ █ ▄ █ █▄▄▄▄▄█
|
||||
|
||||
Manage systemctl --user restart codeman-web · journalctl --user -u codeman-web -f
|
||||
Update re-run the install line, or App Settings → System → Updates
|
||||
Docs https://github.com/Ark0N/Codeman/wiki
|
||||
|
||||
Security: Codeman binds 127.0.0.1. Tailscale authenticates every device before a
|
||||
packet reaches it. Details: docs/security-architecture.md
|
||||
```
|
||||
|
||||
The QR comes from the `qrcode` package Codeman already depends on
|
||||
(`node -e "require('qrcode').toString(url, {type:'terminal', small:true}, …)"` from
|
||||
`$INSTALL_DIR`, verified locally: 17 rows by 45 columns). Skipped when the terminal has no
|
||||
color support or fewer than 50 columns. The QR encodes the plain URL, not an auth token:
|
||||
the tailnet is the login.
|
||||
|
||||
### 3.3 Express mode and flags
|
||||
|
||||
Env vars stay (`CODEMAN_TAILSCALE=1`, `CODEMAN_HOST`, `CODEMAN_PASSWORD`,
|
||||
`CODEMAN_NONINTERACTIVE=1`, `CODEMAN_PORT`). Flags are added because they are
|
||||
discoverable from the one-liner and pipe through `bash -s --`:
|
||||
|
||||
```bash
|
||||
curl -fsSL https://getcodeman.com/install | bash -s -- --tailscale --service
|
||||
curl -fsSL https://getcodeman.com/install | bash -s -- --lan --password 'x' --service
|
||||
curl -fsSL https://getcodeman.com/install | bash -s -- --local --run
|
||||
curl -fsSL https://getcodeman.com/install | bash -s -- --tailscale --name codeman-build --yes
|
||||
```
|
||||
|
||||
| Flag | Meaning |
|
||||
| ---- | ------- |
|
||||
| `--tailscale` / `--lan` / `--local` | Answer 1/3 (same semantics as `CODEMAN_TAILSCALE=1`, `CODEMAN_HOST=0.0.0.0`, `CODEMAN_HOST=127.0.0.1`) |
|
||||
| `--name <n>` / `--no-rename` | Answer 2/3: rename the node to `<n>`, or never ask |
|
||||
| `--service` / `--run` / `--no-start` | Answer 3/3 |
|
||||
| `--yes` | Accept every default, still prompt for a login URL (a human must open it) |
|
||||
| `--password <p>` | Same as `CODEMAN_PASSWORD` |
|
||||
| `--port <n>` | Same as `CODEMAN_PORT`; the serve target follows it |
|
||||
|
||||
`--yes` differs from `CODEMAN_NONINTERACTIVE=1`: it is the interactive user saying "I trust
|
||||
the defaults", so it may install software and may wait on a login URL. Non-interactive stays
|
||||
the CI contract and never installs Tailscale.
|
||||
|
||||
## 4. The Tailscale flow, v2
|
||||
|
||||
The state machine from the previous plan stays; these are the changes.
|
||||
|
||||
1. **Preflight, before the build** (`tailscale_preflight`): installed? -> install
|
||||
(Linux: official script; macOS: brew cask, else download link and wait). Logged in? ->
|
||||
`tailscale up` with the URL printed prominently and a 5-minute poll. Operator (Linux):
|
||||
grant once under the single sudo session. HTTPS certs: poll instead of ask (Ctrl+C
|
||||
during the poll skips Tailscale for this run rather than ending the installer). The
|
||||
rename default does not depend on whether this run performed the login (decided NO
|
||||
everywhere), so nothing records it.
|
||||
2. **Name** (`tailscale_choose_name`, question 2/3): shown only on the Tailscale route.
|
||||
Default `codeman-<oshostname>` sanitized to `[a-z0-9-]`, max 63. Applied with
|
||||
`ts_cmd_serve set --hostname`, then poll `.Self.DNSName` until it carries the new name
|
||||
(up to 60 s). Order matters: this runs before any serve mutation (section 2 trap).
|
||||
Declining keeps the node name. On a re-run against a node already named `codeman-*`,
|
||||
the question is skipped.
|
||||
3. **Serve, after the service is up** (`setup_tailscale_serve`): unchanged idempotent
|
||||
"kept as-is" path first. When `:443` root belongs to another target, the new prompt is:
|
||||
|
||||
```
|
||||
tailscale serve already sends https://tnode.tailf80371.ts.net to port 8080.
|
||||
1) Add Codeman under a path: https://tnode.tailf80371.ts.net/codeman (default)
|
||||
2) Use another port: https://tnode.tailf80371.ts.net:8443
|
||||
3) Replace the existing mapping with Codeman
|
||||
4) Skip Tailscale for now
|
||||
```
|
||||
|
||||
Option 1 writes `--base-url /codeman` into the service unit (it is a `WebLaunchOptions`
|
||||
field already, and `buildWebArgs` carries it) and runs
|
||||
`tailscale serve --bg --set-path /codeman <port>`. Option 2 runs `--https=8443`.
|
||||
`detect_tailscale_serve_url` learns to recognize all three shapes (root, path, port) so
|
||||
uninstall, the security notice and the re-run default keep working.
|
||||
4. **Warm the certificate.** Right after serve is configured, fire one background
|
||||
`curl -sk https://<url>/api/status` so Let's Encrypt issuance overlaps the rest of the
|
||||
install instead of adding 30 s to the verify step.
|
||||
5. **Verify** as today (200 or 401 on `/api/status`), with the path-aware URL.
|
||||
6. **Services hint** (option C): when `.Self.Tags` is non-empty and `serve --help`
|
||||
lists `--service`, the done screen adds one line: "This is a tagged node, so it can also
|
||||
host `https://codeman.<tailnet>.ts.net` as a Tailscale Service: see Remote Access in the
|
||||
wiki." No flow, no prompt.
|
||||
7. **macOS**: the App Store and Standalone variants cannot run before login, so a
|
||||
LaunchAgent plus serve only comes back after someone logs in. The done screen says so on
|
||||
macOS. The Mac mini (`arbbot`, headless, system LaunchDaemon) is the reference for the
|
||||
"headless Mac" caveat, and `install.sh` must keep refusing to replace a LaunchDaemon it
|
||||
did not write (today it removes one; that is a bug for the Mac mini and is fixed here:
|
||||
detect `UserName` in the daemon plist and leave it alone with a message).
|
||||
8. **Uninstall** additionally offers to restore the original node name when this installer
|
||||
renamed it (the original is recorded in `~/.codeman/install.json`, the one marker file
|
||||
this feature adds, because tailscaled does not remember previous names).
|
||||
9. **Subcommands**: `install.sh tailscale` (unchanged purpose, now runs the v2 flow),
|
||||
`install.sh name [<n>]` (rename with the off/rename/re-add dance), `install.sh status`
|
||||
(prints the done screen again, URL and QR included, for the "what was my URL" moment).
|
||||
|
||||
## 5. In-app: the URL stays discoverable
|
||||
|
||||
Small, read-only, and the first server-side code this feature has ever needed.
|
||||
|
||||
- **`GET /api/system/remote-access`** returns
|
||||
`{ tailscale: { installed, connected, dnsName, url, mode: 'root'|'path'|'port'|null } }`
|
||||
by running `tailscale status --json` and `tailscale serve status --json` through
|
||||
`execFile` with the existing exec timeout, cached 30 s, resolved through the same
|
||||
`get_tailscale_path` search as the installer (PATH, then the macOS app bundle), and a
|
||||
no-op under `VITEST` like every other IO probe. Never mutates serve config.
|
||||
- **App Settings -> Remote access** gains a **Tailscale** row above the Cloudflare toggle:
|
||||
the URL as a copy chip, a QR button reusing `showTunnelQR`'s modal, and when nothing is
|
||||
configured a one-line hint with `bash ~/.codeman/app/install.sh tailscale`. The welcome
|
||||
screen's "open on your phone" affordance shows the same QR.
|
||||
- **`codeman doctor`** grows a `tailscale` entry under `other` in
|
||||
`config/dependency-registry.ts`: installed, connected, serving Codeman (URL). Pure
|
||||
engine, injectable probe host, like the existing rows.
|
||||
- No new SSE event, no settings key, no state.json change.
|
||||
|
||||
## 6. Security posture
|
||||
|
||||
Nothing widens. The bind stays loopback; the tailnet is the authentication boundary;
|
||||
`.ts.net` is already in `DEFAULT_TRUSTED_HOST_SUFFIXES`. New surfaces are read-only
|
||||
probes. `install.sh` still never runs `tailscale serve reset`, still touches only the
|
||||
mapping it created, and gains one more never: it never advertises a Tailscale Service or
|
||||
runs `tailscale funnel`. The sudo keep-alive loop is killed by the existing `cleanup` trap.
|
||||
The rename records the previous name locally and offers the reversal at uninstall.
|
||||
|
||||
## 7. Implementation inventory
|
||||
|
||||
| File | Change |
|
||||
| ---- | ------ |
|
||||
| `install.sh` | New `parse_flags`, `preflight_summary`, `ask_everything` (the three questions), `sudo_session`, `run_step` (spinner + log), `tailscale_preflight`, `tailscale_choose_name`, `tailscale_rename_node`, `print_done_screen`, `print_qr`, `status` subcommand, `name` subcommand. Modified: `main` (reordered into ask -> work -> done), `choose_network_binding` (question 1/3, same defaults), `setup_tailscale_serve` (path/port options), `detect_tailscale_serve_url` (three shapes), `setup_systemd_service`/`setup_launchd_service` (`--base-url`, LaunchDaemon guard), `uninstall` (rename reversal), header docs (flags). Removed from the main flow: the cloudflared prompt, the tunnel-service prompt. bash 3.2 rules unchanged. |
|
||||
| `src/web/routes/system-routes.ts` | `GET /api/system/remote-access` |
|
||||
| `src/tailscale-status.ts` (new) | Pure parser for the two JSON shapes + the IO wrapper; unit-tested against captured `serve status --json` fixtures (root, path, port, foreign target, none) |
|
||||
| `src/config/dependency-registry.ts`, `src/utils/dependency-checker.ts` | `tailscale` doctor row |
|
||||
| `src/web/public/index.html`, `settings-ui.js`, `panels-ui.js` | Tailscale row + QR, welcome-screen QR |
|
||||
| `test/install-sh-invariants.test.ts` | Extend: flags documented in the header, no `serve reset`, no `funnel`, no `--service` advertise, every serve mutation goes through `ts_cmd_serve`, rename happens before serve in `main` (static order check) |
|
||||
| `.github/workflows/ci.yml` | The bash 3.2 step additionally sources the script with stubbed `ts_cmd`/`ts_cmd_serve`/`read_reply` and drives `ask_everything` through all three answers and the 443-occupied menu |
|
||||
| `test/tailscale-status.test.ts`, `test/routes/system-routes-remote-access.test.ts` | Parser + route |
|
||||
| Docs | README install + remote-access sections, `docs/wiki/Installation.md`, `Remote-Access.md` (naming options table, Services caveat, path/port variants), `Mobile-Guide.md`, `Running-As-A-Service.md` (macOS login caveat), `FAQ.md`, `docs/security-architecture.md` §A, CLAUDE.md Scripts & Tunnel paragraph, `docs/tailscale-installer-plan.md` gets a pointer here. getcodeman.com copy lives outside the repo (maintainer handbook). |
|
||||
|
||||
Changeset: `minor` (new flags, new subcommands, new API route).
|
||||
|
||||
## 8. Test plan
|
||||
|
||||
Automated (the gate): the static invariants above, the bash 3.2 container drive of the
|
||||
question phase, the JSON parser fixtures, the route test.
|
||||
|
||||
Manual matrix, on a fresh Ubuntu 24 VM and on the Mac mini, since the previous plan's
|
||||
items never ran on a fresh machine:
|
||||
|
||||
1. Tailscale absent, declined -> local-only, done screen shows the retrofit command.
|
||||
2. Tailscale absent, accepted -> install, login URL, operator, certs toggle polled, rename
|
||||
question shown (default no), service, serve, URL verified, QR scans on a phone, PWA installs.
|
||||
3. Tailscale present and logged in on a pre-existing node -> rename default NO, URL is the
|
||||
node name, `serve status` gains exactly one entry.
|
||||
4. `:443` root occupied -> path option -> `https://<node>/codeman` answers, hooks still
|
||||
fire (raw port), `install.sh status` prints the path URL.
|
||||
5. Rename on a node that already has our serve mapping (`install.sh name`) -> off, rename,
|
||||
re-add, `serve status` has no stale key.
|
||||
6. Re-run the one-liner -> quiet update, binding and name preserved, no prompts.
|
||||
7. `--yes` end to end; `CODEMAN_NONINTERACTIVE=1` end to end (no software installed).
|
||||
8. Uninstall -> mapping removed, other mappings intact, rename reversal offered.
|
||||
9. Mac mini: LaunchDaemon left alone with the message; done screen carries the login caveat.
|
||||
|
||||
## 9. Phasing and open decisions
|
||||
|
||||
**Phase 1 (this round):** the reorder, the three questions, one consent + one sudo, flags,
|
||||
the done screen with QR, Tailscale preflight-before-build, the path/port answer for an
|
||||
occupied 443, the rename step, `status` and `name` subcommands, docs.
|
||||
|
||||
**Phase 2:** the in-app Tailscale row + QR, `codeman doctor` row, the `remote-access`
|
||||
route. Independent of phase 1 and useful on its own for existing installs.
|
||||
|
||||
**Phase 3 (optional):** replace the bash service writers with `codeman service install`
|
||||
once that command can carry `CODEMAN_PASSWORD` behind an explicit flag; and a Tailscale
|
||||
Services flow if a tagged-fleet user asks for `codeman.<tailnet>.ts.net`.
|
||||
|
||||
Decisions for the maintainer:
|
||||
|
||||
1. **Rename default.** Decided 2026-09-20: always NO; the yes answer, `--name` and
|
||||
`install.sh name` are the ways in. (The proposal was YES only when this run had joined
|
||||
the tailnet, NO otherwise; rejected because the host is used for other things.)
|
||||
2. **Name pattern.** `codeman-<hostname>` (proposed; unique per machine, and two Codemans
|
||||
on one tailnet stay distinguishable) versus plain `codeman` (nicer once, collides on the
|
||||
second install, Tailscale silently appends `-1`).
|
||||
3. **Path versus port** as the default answer for an occupied 443. Proposed: path, because
|
||||
the URL has no port and `--base-url` already exists for exactly this proxy shape.
|
||||
4. **Whether Phase 2 ships in the same release.** It is the part that helps people who
|
||||
installed months ago.
|
||||
@@ -352,6 +352,177 @@ path but the SESSION (`session.remote`): a remote session never falls back to lo
|
||||
`fs`, and a local session never opens an ssh connection — including for attachment
|
||||
records, which are keyed to the session that registered them.
|
||||
|
||||
## Wake-on-LAN from user input
|
||||
|
||||
A durable remote session survives an SSH drop (COD-104/108), but nothing brought the
|
||||
HOST back. When the remote machine suspended, the local pane's `ssh` child **stalled**
|
||||
rather than exited: `tmux send-keys` SUCCEEDS against a stalled pane, so typed input
|
||||
vanished with no error anywhere, and without a keepalive the pane could look alive for
|
||||
the OS TCP timeout. The only recovery was waiting for the reconnect watcher, which
|
||||
gave up after ~13 minutes and, once exhausted, never retried.
|
||||
|
||||
An **optional** `wakeMac` (one or more MAC addresses, comma-separated) or `wakeCommand` on a
|
||||
remote host closes that: on user input, `POST /api/sessions/:id/input` probes the host, and if
|
||||
it is unreachable it wakes it, polls until the host answers, reattaches the pane
|
||||
(`Session.reattachRemote()`, which idempotently attaches the still-running remote tmux — the
|
||||
agent conversation is not restarted), and flushes the input that arrived meanwhile.
|
||||
Implementation: `src/remote-wake.ts`.
|
||||
|
||||
The same wake path also serves **opening** a session, which is where a sleeping host used to
|
||||
be a dead end: pressing Run on a remote case (`POST /api/quick-start`) or Attach on a
|
||||
discovered remote tmux session (`POST /api/sessions` + `attachRemoteSession`) probes the host
|
||||
first, and on a sleeping one wakes it, waits for SSH and only then runs the tmux prereq probe.
|
||||
Without that the run failed with `could not verify tmux on remote host …` — an ssh error that
|
||||
blames tmux for a machine that is merely suspended. The wait is **blocking** (the caller gets
|
||||
the session or the error) but bounded by `REMOTE_WAKE_REQUEST_READY_TIMEOUT_MS` (40 s) rather
|
||||
than the 90 s session default, because the dashboard sits behind a reverse proxy whose default
|
||||
`proxy_read_timeout` is 60 s: a longer wait would be cut off at the proxy while the session was
|
||||
still being created. The budget covers the whole request, not just the wait (40 s wake + 1.5 s
|
||||
probe + the tmux prereq probe's own 15 s timeout = 56.5 s worst case). A host with no wake target is not even probed on this path, so nothing
|
||||
changes for it, and `remote:hostWaking` is broadcast without a `sessionId` (the toast then reads
|
||||
"the session starts when it is back" — there is no session yet, and no input queued behind it).
|
||||
|
||||
Two wake paths, `wakeCommand` first because it is the explicit override:
|
||||
|
||||
- **`wakeMac`** — Codeman builds the magic packet itself (`buildMagicPacket`, six `0xFF`
|
||||
bytes then the MAC repeated 16×; the shape is asserted byte-for-byte) and broadcasts it
|
||||
over UDP port 9 (`sendWakePackets`). This is the normal case: no external script, and one
|
||||
MAC list per host instead of one per consumer.
|
||||
- **`wakeCommand`** — a single executable path, run WITHOUT a shell. For hosts that need a
|
||||
router/another machine to send the packet.
|
||||
|
||||
**UI**: a banner (`#hostWakeBanner`, `host-wake-ui.js`) appears while the ACTIVE remote
|
||||
session's host is unreachable — amber, since the Codeman session is healthy and only the
|
||||
machine is asleep. With a wake target the action is **Wake** (`POST /api/sessions/:id/wake`);
|
||||
with none it is **Configure WoL** and opens `#wakeConfigModal`, a small form for that host's
|
||||
`wakeMac`/`wakeCommand` that saves with `PUT /api/remote-hosts/:id` (in multi-user mode that
|
||||
GET is admin-only, so a non-admin is told the setting is admin-only instead of "host not
|
||||
found"). Reachability for the banner comes from `GET /api/sessions/:id/reachability`: once
|
||||
when the remote tab is activated (a user action), and every 30 s while the tab is visible
|
||||
**only for a host with a wake target** — each poll is a TCP connect to the host, and a timer
|
||||
that connects to a host Codeman could not wake anyway is exactly the timer-driven traffic
|
||||
the keepalive rule below rejects (it cannot wake a host, but it can keep an activity-based
|
||||
suspend timer from firing). A host the probe cannot reach (see the next section) is never
|
||||
polled. ⚠️ The button is pressed from the SAME
|
||||
dashboard as Run/Attach, so it holds its request open under the same proxy and uses the same
|
||||
40 s budget — and it **queues nothing**: browser keystrokes travel over the WebSocket, which
|
||||
deliberately does not pass through the registry (that is the hot path this feature keeps its
|
||||
hands off), so the banner says "waiting for the host to come back" for the button and only
|
||||
claims "input is queued" when the HTTP input path actually buffered bytes
|
||||
(`queuedInput` on the two SSE events).
|
||||
|
||||
**Hosts behind a jump host or SOCKS proxy are reachability-UNKNOWN.** The probe is a bare
|
||||
TCP connect to `host:port`, and a host reached through `jumpHost`, `socksProxy` or a
|
||||
`ProxyCommand`/`ProxyJump` in `extraSshOptions` does not answer that even while ssh works —
|
||||
the direct address may not route at all (the cloudflared case). Acting on the resulting
|
||||
"unreachable" verdict was wrong three times over: a permanent banner over a healthy session,
|
||||
a create-path error that replaced a genuine "needs tmux" with "not reachable", and — with a
|
||||
wake target configured — every HTTP input buffered for the life of the session, because the
|
||||
readiness poll could never succeed. `isProbeable()` (`remote-wake.ts`) decides from the
|
||||
proxy fields, which travel on `WakeableRemote`; for such a host the registry delivers input
|
||||
unchanged, `GET …/reachability` answers `reachable: null, probeable: false` (unknown is not
|
||||
`false`, and only a proven `false` raises the banner), the create/attach path is not gated
|
||||
(`ensureHostAwake` → `'unprobeable'`, handled like `'no-target'`), and the quick-start
|
||||
"not reachable" message is reserved for a **proven** unreachable host (`=== false`). A wake
|
||||
target can still be fired for it through `POST /api/sessions/:id/wake`, blind: the packet or
|
||||
command goes out and the response says only whether it did — no readiness poll, no reattach
|
||||
(the COD-108 watcher owns the pane once ssh works again), no "waking" toast.
|
||||
|
||||
The invariants worth keeping:
|
||||
|
||||
- **Authorization comes before the wake.** In multi-user mode the attach path
|
||||
(`POST /api/sessions` + `attachRemoteSession`) answers `403` to a non-admin BEFORE the
|
||||
host is looked up or probed: remote hosts are admin-only infrastructure everywhere else
|
||||
(the list is `[]` for a non-admin, write and discovery routes are `adminOnly`), and the
|
||||
wake spawns the host's `wakeCommand` or broadcasts a packet — a gate that came after the
|
||||
wake handed an unprivileged account a way to run that executable for any configured
|
||||
`hostId`, hold the request for the wake budget, and only then be refused for the
|
||||
workingDir. The quick-start path resolves its remote case through `canAccessOwned`
|
||||
first. Pinned in `test/routes/session-remote-wake.test.ts` (wake spy stays empty).
|
||||
- **The caller is told what happened to its bytes.** The non-wait input route answers
|
||||
`{buffered:true}` when the registry took the chunk and `{buffered:true, dropped:true}`
|
||||
when it was over the cap and is gone; the send-and-wait route answers `OPERATION_FAILED`
|
||||
when the host never comes back, like the create and attach paths, instead of writing
|
||||
into the stalled pane and reporting `delivered:true` plus a timeout. Flushed chunks are
|
||||
written with `fromUser`, so a first prompt that was buffered through a wake can still
|
||||
name the tab.
|
||||
- **Only an EXPLICIT request may wake a host:** user input on an established session, the wake
|
||||
button, or the user's own session create/attach request (`ensureHostAwake`). Everything that
|
||||
runs on a TIMER must never wake one — the COD-108 watcher, the server's dropped-session
|
||||
handler, boot recovery and session discovery have no access to the wake registry, and neither
|
||||
has the shared session service, because `cron-service.ts` builds sessions there with nobody
|
||||
waiting on the answer; a wake on such a path would re-wake the host seconds after every
|
||||
suspend, so it could never stay asleep (the same failure `hufflepuff-mcp-lazy` exists to
|
||||
prevent for MCP keepalives). A reachability check, a discovery listing and the tmux prereq
|
||||
probe never wake: they are questions, not actions. All of it is enforced by tests in
|
||||
`test/remote-wake.test.ts` (two wiring guards: one pins the importers — the route module and
|
||||
`server.ts`, which holds the registry for its LIFETIME only, `drop()` on session cleanup and
|
||||
`stop()` on shutdown — and one asserts `server.ts` calls nothing but those two, while
|
||||
`ensureHostAwake` has exactly one caller file) and `test/routes/session-remote-wake.test.ts`,
|
||||
not by comments.
|
||||
- **Detection is a bare TCP connect** to the SSH port (then the configured `port`, else 22),
|
||||
throttled per session, and only for wake-enabled hosts. No `ServerAliveInterval` is added to
|
||||
the launch command: keepalives push bytes into an otherwise idle connection every interval,
|
||||
which is exactly what a byte-threshold idle detector must not count as activity. A probe is
|
||||
~200 bytes per 30 s, orders of magnitude below any such threshold, and the SYN alone cannot
|
||||
wake a host.
|
||||
- **Input is buffered while a wake is in flight** (`REMOTE_WAKE_PENDING_MAX_BYTES`,
|
||||
oldest whole chunks dropped, bounded so user input cannot grow memory) and flushed in
|
||||
order after the reattach, with a settle delay so bytes cannot land in a still-connecting
|
||||
pane. ⚠️ A chunk LARGER than the cap (one big paste is one `input` value) is dropped
|
||||
**outright**, never trimmed: it was never typed character by character, so its tail is not
|
||||
"what the user just typed" but a fragment of a command they never sent — the drop is logged
|
||||
instead. ⚠️ Only the HTTP input route reaches the registry; the **WebSocket keystroke path
|
||||
is deliberately NOT wake-aware**, so typing into a sleeping host sends nothing and queues
|
||||
nothing (the banner's Wake button is the recovery for that case, which is why it must not
|
||||
promise queued input). The **send-and-wait** path blocks on the wake instead — its response
|
||||
is open anyway, and buffering would break the wait contract. ⚠️ A flush write that FAILS
|
||||
drops the whole remaining buffer (logged) rather than retaining it: the wake still resolves
|
||||
and marks the host reachable, so the next input takes the deliver path while a retained
|
||||
chunk would wait for the NEXT wake — replayed hours later, after everything typed since,
|
||||
possibly ending in a carriage return. Same policy as the oversized paste.
|
||||
- **The command runs without a shell** (`spawn(path, [], { stdio: 'ignore' })` — `shell`
|
||||
defaults to `false`), the schema
|
||||
requires a single executable path (no arguments, no `$`/backtick), and `wakeMac` is a
|
||||
structural hex-pair allowlist. A broken or missing wake target fails the wake, never the
|
||||
input route.
|
||||
- **`wakeMac`/`wakeCommand` are host-level config, refreshed on recovery AND live**
|
||||
(`rehydrateRemoteHostFields` in `src/remote-hosts.ts` plus `RemoteWakeDeps.resolveRemote`).
|
||||
A session's `remote` block is persisted at launch time, so a field added to
|
||||
`remote-hosts.json` later would otherwise never reach an already-running session — not even
|
||||
across a Codeman restart, and certainly not right after saving the banner's config dialog.
|
||||
Recovery rehydration covers restarts, the (throttled, cache-backed) resolver covers the live
|
||||
session; the host config is authoritative for both (removing the field disables the feature
|
||||
again). Other host-level fields deliberately stay as persisted, so neither path can
|
||||
silently re-point an existing pane's SSH options.
|
||||
- **UI/SSE**: `remote:hostWaking` and `remote:hostWakeFailed` (plus the reused
|
||||
`remote:sessionReconnected`) drive the banner and toasts, all from `host-wake-ui.js` —
|
||||
its handlers are the ONLY definitions, since a second one in another mixin would be
|
||||
silently shadowed by script order. Both carry `queuedInput`, which is true only when the
|
||||
server actually holds bytes for that session — the wording keys off that, not off "a wake
|
||||
is running", so the button path never claims input is queued. In multi-user mode the
|
||||
whole `remote:` family is **session-scoped** (`deriveSseHint`, `server.ts`): an event with
|
||||
a `sessionId` reaches that session's owner, and the create/attach wake — which has no
|
||||
session yet — carries the requesting `username` instead (`ensureHostAwake({ requestedBy })`),
|
||||
since its payload names a `hostId`/`label` that `GET /api/remote-hosts` withholds from
|
||||
non-admins. With neither, it reaches admins only.
|
||||
- **No real IO under vitest.** `probeRemoteHostReachable`, `runRemoteWakeCommand` and the
|
||||
default UDP socket of `sendWakePackets` throw under `VITEST` (as `remote-files.ts` does),
|
||||
so a test that reaches the defaults fails loudly instead of connecting, spawning or
|
||||
broadcasting from CI. Every consumer injects its IO (`RemoteWakeDeps`, the socket
|
||||
factory); `createDefaultRemoteWakeDeps({ probe })` also polls readiness with THAT probe,
|
||||
which is the leak the guard found.
|
||||
|
||||
Tests: `test/remote-wake.test.ts` (decision/throttle table, single-flight registry,
|
||||
buffering + flush order, MAC parsing/magic packet, live host-config resolution, the proxied
|
||||
host, SSE payload routing, the vitest IO guard, and the wiring guard),
|
||||
`test/routes/session-remote-wake.test.ts` (the input route buffers instead of writing into a
|
||||
sleeping host — and writes straight into a proxied one —, the reachability route never wakes
|
||||
and reports a proxied host as unknown, and the wake route reports the no-target case the UI
|
||||
turns into "configure WoL"), `test/sse-routing-remote.test.ts` (multi-user routing of the
|
||||
`remote:` family) and `test/host-wake-banner.test.ts` (banner visibility and when the poller
|
||||
may connect).
|
||||
|
||||
## API
|
||||
|
||||
Routes are registered in `src/web/routes/case-routes.ts`:
|
||||
@@ -365,6 +536,10 @@ Routes are registered in `src/web/routes/case-routes.ts`:
|
||||
| `GET` | `/api/remote-hosts/:hostId/sessions` | Discover `codeman-*` sessions on the host (COD-105; `listRemoteCodemanSessions`, never errors) |
|
||||
| `POST` | `/api/cases/remote-link` | Link a case to a remote host (creates the `RemoteCase`) |
|
||||
|
||||
`RemoteHost` accepts the optional `wakeMac` (magic packet, sent by Codeman) and `wakeCommand`
|
||||
(single executable path, run without a shell, takes precedence) — see **Wake-on-LAN from user
|
||||
input** above.
|
||||
|
||||
Attaching to a discovered session is a **session-create** path, not a host route:
|
||||
`POST /api/sessions` accepts `attachRemoteSession: { hostId, remoteSessionName }`
|
||||
(schema in `schemas.ts`; `remoteSessionName` must match `^codeman-[a-zA-Z0-9._-]+$`),
|
||||
|
||||
@@ -270,9 +270,15 @@ tailscale serve --bg 3000 # HTTPS at https://<node>.<tailnet>.ts.net
|
||||
Only devices on your tailnet can reach it; Tailscale handles identity and
|
||||
terminates TLS with a real Let's Encrypt certificate (so PWA install and web
|
||||
push work). No app password and no `0.0.0.0` bind required. (This is the
|
||||
maintainer's production setup.) `CODEMAN_TAILSCALE=1` presets the choice for
|
||||
automation; the installer never runs `tailscale serve reset` and never touches
|
||||
serve mappings other than `443 -> Codeman's port`.
|
||||
maintainer's production setup.) `CODEMAN_TAILSCALE=1` or `--tailscale` presets
|
||||
the choice for automation. When `:443` on the node already belongs to another
|
||||
app, the installer mounts Codeman under `/codeman` (`tailscale serve --set-path`
|
||||
plus `--base-url`, which keeps the loopback bind and the same host guard) or on a
|
||||
second port rather than replacing it. The installer never runs `tailscale serve
|
||||
reset`, never touches serve mappings other than the one it created, never opens a
|
||||
`tailscale funnel` (public internet, a different risk class) and never advertises
|
||||
a Tailscale Service. Renaming the node (`--name`, `install.sh name`) is opt-in
|
||||
and defaults to no, because the tailnet name is also the machine's SSH identity.
|
||||
|
||||
### B. Authenticated cloudflared tunnel + password
|
||||
|
||||
|
||||
@@ -1,5 +1,10 @@
|
||||
# Tailscale Setup in the Installer (Plan)
|
||||
|
||||
> Superseded in part by [`installer-v2-plan.md`](installer-v2-plan.md) (2026-09-20), which
|
||||
> moved every human step before the build, added the sub-path / second-port answer for an
|
||||
> occupied `:443`, the opt-in rename, flags, and the done screen with a QR code. The
|
||||
> state machine and safety rules below still hold.
|
||||
|
||||
Goal: make "Codeman over Tailscale, with real HTTPS" a first-class, guided path in
|
||||
`install.sh`, instead of a one-line hint pointing at the docs. Today the safest
|
||||
recommended deployment (loopback bind + `tailscale serve`) is exactly what the
|
||||
|
||||
+92
-11
@@ -1,9 +1,9 @@
|
||||
# Agent CLIs
|
||||
|
||||
Codeman drives seven run modes: six agent CLIs plus a plain shell. This page covers picking
|
||||
Codeman drives ten run modes: nine agent CLIs plus a plain shell. This page covers picking
|
||||
one, setting it up, and the differences that actually change how you work.
|
||||
|
||||
## The seven modes
|
||||
## The ten modes
|
||||
|
||||
| Mode | CLI | Get it |
|
||||
| -------------------- | ---------------------------- | ---------------------------------------------------------------------- |
|
||||
@@ -13,6 +13,9 @@ one, setting it up, and the differences that actually change how you work.
|
||||
| **Gemini** | `gemini` | [github.com/google-gemini/gemini-cli](https://github.com/google-gemini/gemini-cli) |
|
||||
| **Antigravity** | `agy` | [antigravity.google](https://antigravity.google) |
|
||||
| **Pi** | `pi` | [pi.dev](https://pi.dev) |
|
||||
| **Grok Build** | `grok` | [github.com/xai-org/grok-build](https://github.com/xai-org/grok-build) |
|
||||
| **DeepSeek Harness** | `dsh` | [github.com/deepseek-ai/deepseek-harness](https://github.com/deepseek-ai/deepseek-harness) |
|
||||
| **OMP** | `omp` | [github.com/can1357/oh-my-pi](https://github.com/can1357/oh-my-pi) |
|
||||
| **Terminal / Shell** | your `$SHELL` | Already installed. |
|
||||
|
||||
Any combination works, including all of them. The run mode is chosen per session from the
|
||||
@@ -47,8 +50,12 @@ If a CLI is installed but a Run button for it never appears:
|
||||
precisely to avoid this; a hand-written plist or unit will not.
|
||||
3. Restart the server after installing a new CLI.
|
||||
|
||||
`pi` is additionally version-probed rather than trusted by name, because `pi` is a generic
|
||||
enough command that something else on your PATH may answer to it.
|
||||
`pi`, `grok`, `omp` and `dsh` are additionally identity-probed rather than trusted by name:
|
||||
`pi` and `omp` are generic enough that something else on your PATH may answer to them,
|
||||
`grok` has npm squatters, and Debian ships an unrelated `dsh` (dancer's shell). Each has a
|
||||
status endpoint (`/api/grok/status`, `/api/deepseek/status`, `/api/omp/status`) that reports
|
||||
the path and version that actually resolved, so a misresolution is visible rather than
|
||||
presenting as "the mode just does not work".
|
||||
|
||||
## Claude is the reference mode
|
||||
|
||||
@@ -62,15 +69,15 @@ output. The other CLIs expose no equivalent.
|
||||
| Respawn cycling and unattended runs | Yes | Yes |
|
||||
| Cron jobs | Yes | Yes |
|
||||
| Docker cases, remote SSH cases | Yes | Yes |
|
||||
| Precise idle detection (hook-driven) | Yes | Output-stabilization fallback, coarser |
|
||||
| Precise idle detection | Yes | Codex: same screen check, via its own prompt and working line. DeepSeek: reports its state itself. Others: output stabilization, coarser |
|
||||
| Auto-resume when a usage limit resets | Yes | No |
|
||||
| Plan usage chip | Yes | No |
|
||||
| Approvals Inbox | Yes | No |
|
||||
| Approvals Inbox | Yes | DeepSeek yes; others no |
|
||||
| Read My Mind | Yes | No |
|
||||
| Ralph loop and its task tracker | Yes | No |
|
||||
| Subagent and team windows | Yes | No |
|
||||
| Model, effort, and ultracode controls | Yes | No |
|
||||
| `stop` and `blocked` wait signals | Yes | 400 if you ask for them explicitly |
|
||||
| `stop` and `blocked` wait signals | Yes | DeepSeek yes; elsewhere 400 if you ask for them explicitly |
|
||||
| The bundled agent skill | Yes | No |
|
||||
|
||||
Everything that makes a session a session works everywhere. What is Claude-only is mostly
|
||||
@@ -124,6 +131,11 @@ Two behaviours that are deliberate and worth knowing:
|
||||
- **The wheel is not forwarded** into its transcript. Codex ignores the mouse reports
|
||||
Codeman would send, so forwarding produced a dead wheel. Scrolling in a Codex session is
|
||||
local scrollback.
|
||||
- **Work detection is Codex's own.** Codex declares its `›` composer glyph and its
|
||||
`esc to interrupt` working line, so it gets the same screen-checked idle detection Claude
|
||||
does; before 1.26.1 every Codex session reported idle for its whole life. Codex
|
||||
conversations also appear in Past Sessions and can be resumed, and on phones the keyboard
|
||||
bar grows `⇧←` / `⇧→` for Codex's queued-message editing and prompt stack.
|
||||
|
||||
### Gemini
|
||||
|
||||
@@ -157,6 +169,60 @@ Pi needs the opposite instincts from every other CLI here.
|
||||
|
||||
Guide: [`docs/pi-integration.md`](https://github.com/Ark0N/Codeman/blob/master/docs/pi-integration.md).
|
||||
|
||||
### Grok Build
|
||||
|
||||
xAI's `grok`, installed with `curl -fsSL https://x.ai/cli/install.sh | bash` into
|
||||
`~/.grok/bin`. Codex-shaped on permissions and OpenCode-shaped on rendering:
|
||||
|
||||
- **Its bypass switch is `--always-approve`**, Grok's own `bypassPermissions` mode, and the
|
||||
Run button sends it the way it sends Codex's. In multi-user mode a user without a grant
|
||||
has it stripped.
|
||||
- **Authentication is Grok's own**: browser OAuth on first run (a device-code screen inside
|
||||
a Codeman pane), `grok login --device-auth` for headless hosts, or `XAI_API_KEY` as a
|
||||
per-session environment override.
|
||||
- It renders a full-screen TUI, so scrolling is local scrollback.
|
||||
|
||||
Guide: [`docs/grok-integration.md`](https://github.com/Ark0N/Codeman/blob/master/docs/grok-integration.md).
|
||||
|
||||
### DeepSeek Harness
|
||||
|
||||
The mode wired least like the others, for two reasons worth knowing before you use it.
|
||||
|
||||
**`dsh` is a launcher, not an agent.** It boots a *profile*, and the three DeepSeek ships
|
||||
(`web`, `headless`, `base`) cannot drive a terminal pane. So "installed" and "runnable" are
|
||||
different questions: the Run menu offers **DeepSeek** only once a pane-capable profile
|
||||
exists, and until then shows **DeepSeek — add a terminal profile…**, which installs the
|
||||
community `dsh-tui` with one click (`pnpm` must be on PATH, because the launcher spawns it
|
||||
directly).
|
||||
|
||||
**Permissions are an environment variable, not a flag.** The harness has no
|
||||
skip-permissions switch. `DSH_PERMISSION_MODE` (`read-only`, `workspace-write`,
|
||||
`danger-full-access`) is the whole control, and it is the one setting Codeman deliberately
|
||||
carries as an environment variable, because the harness reads it as a soft boot-time
|
||||
default. In multi-user mode a user without a grant is clamped to `workspace-write`.
|
||||
|
||||
The reward for the odd wiring: **DeepSeek is the one non-Claude mode with real signals.**
|
||||
Its terminal front door reports idle, working and blocked to Codeman, so a DeepSeek
|
||||
session gets precise idle detection, the `stop` and `blocked` wait signals, and Approvals
|
||||
Inbox items. Answers are read from the harness's own transcript on disk rather than
|
||||
scraped off the pane. The model is not a session setting; it is part of the profile.
|
||||
|
||||
Guide: [`docs/deepseek-integration.md`](https://github.com/Ark0N/Codeman/blob/master/docs/deepseek-integration.md).
|
||||
|
||||
### OMP
|
||||
|
||||
Oh My Pi, installed with `curl -fsSL https://omp.sh/install | sh` into `~/.local/bin`.
|
||||
OMP owns its auth, provider routing and approval mode entirely in `~/.omp`: there is no
|
||||
Codeman-side login, key field, or bypass switch. Run `omp` once outside Codeman to finish
|
||||
its own onboarding, and every session started through Codeman inherits that config. Its
|
||||
documented default approval mode is `yolo`, so an OMP pane auto-approves tool use with no
|
||||
flag from Codeman; change that in OMP's own config, not here.
|
||||
|
||||
OMP conversations appear in Past Sessions and can be resumed, and a respawn continues the
|
||||
same conversation with `--continue`.
|
||||
|
||||
Guide: [`docs/omp-integration.md`](https://github.com/Ark0N/Codeman/blob/master/docs/omp-integration.md).
|
||||
|
||||
### Terminal / Shell
|
||||
|
||||
A plain shell in a tmux session. No agent, no hooks, no idle detection.
|
||||
@@ -180,9 +246,15 @@ respawns. Which variables are accepted depends on the mode:
|
||||
| Gemini | `GEMINI_*`, `GOOGLE_*` |
|
||||
| Antigravity | `ANTIGRAVITY_*` |
|
||||
| Pi | `PI_*` |
|
||||
| Grok | `GROK_*`, `XAI_*` |
|
||||
| DeepSeek | `DSH_*`, `DEEPSEEK_*` |
|
||||
| OMP | `OMP_*` |
|
||||
|
||||
Anything outside the allowlist is rejected at the schema. This is intentional: the allowlist
|
||||
is one global list, so widening it for one CLI widens it for all of them.
|
||||
is one global list, so widening it for one CLI widens it for all of them. In multi-user mode
|
||||
the keys that could redirect a CLI's traffic or move its config home (`DSH_PERMISSION_MODE`,
|
||||
`DSH_HOME`, `DEEPSEEK_BASE_URL`, `OMP_AUTH_BROKER_URL`, and the base URLs and config
|
||||
directories of the others) are dropped for a user without the bypass grant.
|
||||
|
||||
Two things that deliberately do **not** travel as environment variables: **effort**, because
|
||||
an environment variable hard-locks it and blocks `/effort`, and **model**, which is written
|
||||
@@ -192,16 +264,25 @@ into the case's `.claude/settings.local.json` so that `/model` keeps working.
|
||||
|
||||
- **Claude Code** if you want every Codeman feature. Unattended overnight runs, usage-limit
|
||||
auto-resume, the Approvals Inbox, and subagent visualization all assume it.
|
||||
- **Codex, OpenCode, Gemini, Antigravity** when you prefer that agent or that model. You get
|
||||
the session layer, respawn, cron, Docker, and remote SSH; you do not get the hook-driven
|
||||
features.
|
||||
- **Codex, OpenCode, Gemini, Antigravity, Grok, OMP** when you prefer that agent or that
|
||||
model. You get the session layer, respawn, cron, Docker, and remote SSH; you do not get the
|
||||
hook-driven features.
|
||||
- **DeepSeek Harness** if you want DeepSeek's models with real status signals. It is the one
|
||||
non-Claude mode that reports idle, working and blocked to Codeman itself.
|
||||
- **Pi** if you want a fast, unsandboxed agent and you understand what project trust does.
|
||||
- **Shell** for the times you want a terminal on your phone with no agent at all. It is a
|
||||
genuinely useful mode, not a fallback.
|
||||
|
||||
## Pointing one at your own server
|
||||
|
||||
Most of these harnesses can also run against a custom OpenAI-compatible endpoint instead of
|
||||
their native cloud backend, for one session at a time, an opt-in feature covered in full on
|
||||
[Custom Model Endpoints](Custom-Model-Endpoints).
|
||||
|
||||
## Read next
|
||||
|
||||
- [Core Concepts](Core-Concepts) - run modes versus location overlays.
|
||||
- [Custom Model Endpoints](Custom-Model-Endpoints) - run a harness against your own server.
|
||||
- [Settings Reference](Settings-Reference) - model, effort, and permission-mode settings.
|
||||
- [Keeping Agents Running](Keeping-Agents-Running) - what idle detection does per mode.
|
||||
- [Security](Security) - what skipping permission prompts actually means.
|
||||
|
||||
@@ -108,7 +108,7 @@ Conventions for wiki pages:
|
||||
- Images are referenced from the main repository over raw URLs rather than being copied into
|
||||
the wiki.
|
||||
- Say what the default is, especially when it is off. Most of Codeman is opt-in.
|
||||
- Label Claude-only behaviour every time it appears. Six of the seven run modes are not
|
||||
- Label Claude-only behaviour every time it appears. Nine of the ten run modes are not
|
||||
Claude.
|
||||
|
||||
## Conduct
|
||||
|
||||
@@ -50,10 +50,10 @@ A session carries state the case does not:
|
||||
## Run mode
|
||||
|
||||
The **run mode** is which CLI the session runs: `claude`, `opencode`, `codex`, `gemini`,
|
||||
`antigravity`, `pi`, or `shell`. It is chosen at start and does not change afterwards; to
|
||||
`antigravity`, `pi`, `grok`, `deepseek`, `omp`, or `shell`. It is chosen at start and does not change afterwards; to
|
||||
switch, start another session.
|
||||
|
||||
Claude is the reference mode. Six of the seven are not Claude, and a number of Codeman
|
||||
Claude is the reference mode. Nine of the ten are not Claude, and a number of Codeman
|
||||
features are Claude-only for structural reasons rather than missing effort: they depend on
|
||||
Claude Code's hook system or on parsing its terminal output. Every such feature is labelled
|
||||
Claude-only where it appears, and [Agent CLIs](Agent-CLIs) lists them in one place.
|
||||
@@ -68,8 +68,8 @@ Where a case runs is **separate from** which CLI it runs. There are three locati
|
||||
| **Docker** | One long-lived container per case; sessions `docker exec` into it. See [Docker Cases](Docker-Cases). |
|
||||
| **Remote SSH** | A durable tmux server on the remote host, fronted by a local pane running `ssh`. See [Remote SSH Sessions](Remote-SSH-Sessions). |
|
||||
|
||||
This matters because it is a common source of confusion: Docker is **not** an eighth run
|
||||
mode. All seven run modes work in all three locations. A case is docker-backed or
|
||||
This matters because it is a common source of confusion: Docker is **not** an eleventh run
|
||||
mode. All ten run modes work in all three locations. A case is docker-backed or
|
||||
ssh-backed; a session is claude or codex or shell.
|
||||
|
||||
**Web tabs** are the other thing that is not a session. A saved dashboard URL renders as a
|
||||
@@ -155,9 +155,11 @@ report events back: a permission prompt appeared, the turn finished, the agent w
|
||||
task completed. Those events drive tab alerts, the Approvals Inbox, notifications, and the
|
||||
wait primitives.
|
||||
|
||||
This is why some features are Claude-only. The other CLIs have no equivalent hook system,
|
||||
so for them Codeman falls back to watching terminal output, which is coarser: it can see
|
||||
that something happened, not what it was.
|
||||
This is why some features are Claude-only. The one partial exception is DeepSeek Harness,
|
||||
whose terminal front door reports idle, working and blocked to Codeman over the harness's
|
||||
own supervisor contract, so it gets the hook-driven signals without a hook file. The other
|
||||
CLIs have no equivalent, so for them Codeman falls back to watching terminal output, which
|
||||
is coarser: it can see that something happened, not what it was.
|
||||
|
||||
See [Hooks And Integrations](Hooks-And-Integrations).
|
||||
|
||||
@@ -167,7 +169,7 @@ See [Hooks And Integrations](Hooks-And-Integrations).
|
||||
| --------------- | ---------------------------------------------------------------------------- |
|
||||
| **Case** | Named working directory. |
|
||||
| **Session** | One CLI in one tmux session. |
|
||||
| **Run mode** | Which CLI: claude, opencode, codex, gemini, antigravity, pi, shell. |
|
||||
| **Run mode** | Which CLI: claude, opencode, codex, gemini, antigravity, pi, grok, deepseek, omp, shell. |
|
||||
| **Respawn** | Restarting the CLI on idle to keep an unattended run going. |
|
||||
| **Ralph loop** | An autonomous single-session task loop. |
|
||||
| **Orchestrator**| A phased plan driven across multiple agents. |
|
||||
@@ -178,6 +180,6 @@ See [Hooks And Integrations](Hooks-And-Integrations).
|
||||
## Read next
|
||||
|
||||
- [The Dashboard](The-Dashboard) - what the UI is showing you.
|
||||
- [Agent CLIs](Agent-CLIs) - the seven run modes in detail.
|
||||
- [Agent CLIs](Agent-CLIs) - the ten run modes in detail.
|
||||
- [Keeping Agents Running](Keeping-Agents-Running) - respawn, idle detection, usage limits.
|
||||
- [`docs/architecture-invariants.md`](https://github.com/Ark0N/Codeman/blob/master/docs/architecture-invariants.md) - the mechanisms behind all of this, for contributors.
|
||||
|
||||
@@ -0,0 +1,183 @@
|
||||
# Custom Model Endpoints
|
||||
|
||||
Point a harness at your own OpenAI-compatible server instead of its native cloud backend, for
|
||||
one session at a time. "Custom endpoint" covers **local** hardware (llama.cpp, Ollama, vLLM,
|
||||
a home GPU rig, DGX Spark, Strix Halo) and **cloud** services (Azure AI Foundry's
|
||||
OpenAI-compatible endpoint, OpenRouter, a company gateway) alike, anything answering
|
||||
`GET /v1/models` and `POST /v1/chat/completions` in the standard shape.
|
||||
|
||||
**Off by default.** Turn it on in App Settings → Models → **Custom model endpoints**.
|
||||
|
||||
## Adding an endpoint
|
||||
|
||||
Still in App Settings → Models → Custom model endpoints:
|
||||
|
||||
1. **+ Add endpoint** — give it an id, a label, and the base URL (`http://192.168.1.50:8080`,
|
||||
say). An API key is optional; most local servers don't check one.
|
||||
2. **Discover** — fetches the endpoint's own model list over `GET /v1/models` and stores it.
|
||||
3. Pick a **default model** from what was discovered. This is the model the Run-menu entry
|
||||
applies directly when only one model is discovered; with two or more, it's just the one
|
||||
pre-marked in the picker dialog described below, not a silent default.
|
||||
|
||||
Endpoint management is admin-only in multi-user mode, the same as remote hosts and Docker
|
||||
hosts — these are machine-level infra, not a per-user setting.
|
||||
|
||||
**Model lists refresh themselves.** Every saved endpoint is re-discovered automatically every
|
||||
5 minutes in the background, so a model the server starts serving later — or stops serving —
|
||||
shows up without another manual click of **Discover**. One endpoint being unreachable on a
|
||||
given cycle (powered off, wrong network) never blocks the others from refreshing.
|
||||
|
||||
**Context length is picked up automatically where it can be, safely.** Against a
|
||||
llama.cpp/llama-swap server, discovery also learns each _currently loaded_ model's real
|
||||
context window and applies it to the launched session (Claude Code today — see below), so
|
||||
the harness stops assuming a large default window for a model name it doesn't recognise and
|
||||
overflowing a much smaller real one. It's deliberately never probed for a model that isn't
|
||||
already loaded, since asking a llama-swap server about an unloaded model can trigger an
|
||||
actual, slow model swap as a side effect — a model just not currently loaded keeps whatever
|
||||
context length an earlier cycle already learned for it instead.
|
||||
|
||||
## Running a session against one
|
||||
|
||||
With the setting on and at least one endpoint carrying a discovered model, the **Run**
|
||||
dropdown grows a **Custom Endpoints** section: one entry per harness that can redirect to a
|
||||
custom endpoint, per saved endpoint, e.g. "Claude Code (llama.cpp)". Picking one starts a
|
||||
session on that harness exactly the way its own entry would. It is a one-off "try this
|
||||
endpoint" action, not a sticky mode — the plain **Run** button still means "this harness,
|
||||
native cloud" afterward, and a fresh session never inherits whatever the last one was
|
||||
pointed at.
|
||||
|
||||
**Which model it uses depends on how many the endpoint has discovered.** With exactly one,
|
||||
the session launches straight away on that model — nothing to choose. With two or more, a
|
||||
small dialog asks which one to use for this launch before starting the session; the
|
||||
endpoint's default model, if set, is marked but not auto-picked, so a launch can deliberately
|
||||
use a different one without changing the saved default.
|
||||
|
||||
**For opencode, Codex, Gemini, Pi, Grok, DeepSeek and OMP, picking an entry launches
|
||||
straight onto the endpoint** — no restart, because the endpoint is applied before the
|
||||
session's process ever starts. **Claude still restarts the harness's process in place** —
|
||||
same tab, same conversation (`--resume`) — after a normal native launch, since that restart
|
||||
is far less jarring for Claude than for the other seven, whose own TUI can fully
|
||||
reinitialize on a restart. Either way, every supported harness reads its endpoint config at
|
||||
process start, never per turn, so there is no live hot-swap while a turn is running.
|
||||
|
||||
Picking an entry that launches a **brand-new** Claude session waits (up to 20 seconds) for it to
|
||||
finish its own startup before applying — a freshly started CLI reports itself as busy for its
|
||||
boot sequence, and applying to a genuinely busy session is refused so a real, in-progress
|
||||
turn is never interrupted out from under you. A session that is still busy after that wait
|
||||
(a very slow-starting CLI, or one you started typing into right away) surfaces that refusal
|
||||
as an ordinary error, which now stays on screen with a close button instead of vanishing
|
||||
after a few seconds — read it, it names the actual reason rather than a generic failure.
|
||||
|
||||
Entries are hidden entirely for a session in a **remote (SSH) or Docker case** — support for
|
||||
redirecting those hasn't landed yet, see below. The picker also only appears in the desktop
|
||||
**Run** dropdown; the phone home screen builds its own run picker separately and does not
|
||||
currently offer these entries.
|
||||
|
||||
**Against llama-swap, applying a selection also starts the actual model load, rather than
|
||||
waiting on your first prompt to do it.** llama-swap has no "switch model" button of its own
|
||||
— the only thing that starts a swap is a real request naming the model, and confirmed live:
|
||||
just applying a selection never reached llama-swap's own logs at all until something asked
|
||||
it to load. Picking an entry now also sends the smallest real request that will trigger
|
||||
that load, in the background, the moment the target model isn't already loaded and ready.
|
||||
|
||||
**The centred loading banner has no countdown and no automatic timeout — it waits as long as
|
||||
it takes, and tells you so.** When it knows the model's discovered file size (its GB figure,
|
||||
when llama-swap states one) it's shown too, e.g. "Loading qwen3.8-27b (16.4 GB) on
|
||||
llama-swap — this can take a while depending on your hardware and the model size." An
|
||||
earlier version tried to estimate and enforce a time limit, but real load time depends on
|
||||
hardware this feature has no way to know, so a fixed number was always a guess — worse, one
|
||||
that could kill a genuinely slow load partway through. If it really is taking too long, a
|
||||
**Cancel** button right on the banner ends the wait and **closes the session that load was
|
||||
for**, on your own call rather than a guessed deadline.
|
||||
|
||||
**The banner also shows a real, live second line of what llama.cpp itself is doing** — not
|
||||
a made-up progress phase, the actual next line the `llama-server` process printed, e.g.
|
||||
"llama.cpp: load_model: loading model '/models/.../Qwen3.8-27B.gguf'" then later
|
||||
"llama.cpp: llama_server: model loaded". It comes straight from llama-swap's own event
|
||||
feed, filtered down to just the backend process's own output (not llama-swap's own request
|
||||
logging), and stays on whatever it last said once the load goes quiet, rather than
|
||||
clearing back to nothing.
|
||||
|
||||
**You'll also be told if a session's model gets swapped out from under it later, not just
|
||||
at launch.** The conflict warning above only fires at the moment you launch or apply a
|
||||
model — llama.cpp only runs one model at a time, so if a DIFFERENT session using the same
|
||||
endpoint later triggers its own load, whatever was loaded before (including a session you
|
||||
already had running) gets silently evicted, with no warning at that instant since nothing
|
||||
conflicted when it was first set up. A background check (every 20 seconds) catches this
|
||||
after the fact and shows a toast naming which session lost its model and what's loaded now
|
||||
— so you know before typing into that session that it's about to reload (and, in turn,
|
||||
evict whatever displaced it).
|
||||
|
||||
**Claude Code specifically gets three extra fixes applied automatically:**
|
||||
|
||||
- Its discovered context length (see above) is passed through as
|
||||
`CLAUDE_CODE_MAX_CONTEXT_TOKENS`, so it doesn't send a full-size prompt against a much
|
||||
smaller real local context and overflow it.
|
||||
- Its session runs with an isolated `CLAUDE_CONFIG_DIR`, so the injected API key never sits
|
||||
in the same directory as a stored claude.ai login — that combination is harmless for actual
|
||||
requests (the API key wins) but the CLI still prints a "both claude.ai and
|
||||
ANTHROPIC_API_KEY set" warning about it, which this avoids entirely. The isolated directory
|
||||
keeps a link back to your real session history so the response viewer and similar features
|
||||
still work for that session. That isolated directory starts with no prior approvals of its
|
||||
own, so Codeman also pre-approves the injected key the same way answering Claude Code's own
|
||||
"Detected a custom API key" prompt once would — without it, that prompt would otherwise
|
||||
reappear on every single launch with nobody there to answer it.
|
||||
- **That same fresh isolated directory also looks like a brand-new Claude Code profile**, so
|
||||
without this fix it replayed the WHOLE first-run sequence every single launch: the theme
|
||||
picker, the security-notes screen, the "trust this folder?" dialog, and a one-time warning
|
||||
about running with permissions bypassed — none of which a real, already-used profile shows
|
||||
again. Codeman now pre-seeds that same "already been through this once" state (onboarding
|
||||
completed, this session's own project marked trusted, the bypass-permissions warning
|
||||
acknowledged) so a custom-model launch reaches the actual conversation exactly as fast as a
|
||||
native cloud one does, instead of stopping at a wizard with nobody there to click through it.
|
||||
|
||||
**If a model's real context is too small for Claude Code to even get started, you get a
|
||||
warning instead of a confusing failure.** Claude Code's own system prompt and tools take up
|
||||
roughly 40K tokens on their own, before you've typed anything — a small local model with a
|
||||
smaller real context than that fails outright on the very first message, no matter what
|
||||
context size Codeman tells it to expect (raising the declared context only changes when
|
||||
Claude Code trims _conversation history_, and there is none yet on message one). Picking
|
||||
such a model now shows an in-app dialog naming the model, its discovered context and what's
|
||||
needed, before anything launches or restarts, with the fix spelled out: reconfigure
|
||||
llama-swap to give that model (or a smaller one) an explicit larger context instead of
|
||||
relying on auto-fit (`--fit-ctx`), which sizes the context around fitting the biggest model
|
||||
rather than the biggest context — for example adding `-c 65536` to that model's llama-swap
|
||||
entry. "Launch anyway" is still there if you want to try regardless.
|
||||
|
||||
## Which harnesses actually work
|
||||
|
||||
| Harness | Status |
|
||||
| ---------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| **Claude Code, opencode, Pi, Grok, OMP** | Verified end-to-end against a real local server. |
|
||||
| **Codex** | Config is correct, and plain chat can work against a server that speaks the Responses API — but a real tool-call attempt comes back as inert text instead of running, so it's still not usable for real coding work. |
|
||||
| **Gemini** | Fails with an auth error gemini-cli raises once redirected. Unresolved; don't rely on it yet. |
|
||||
| **DeepSeek** | The original 404 is root-caused and fixed (DeepSeek Harness's own code was missing a `/v1` most local servers require) — not yet re-run against a real `dsh` install to confirm end-to-end. |
|
||||
| **Antigravity** | No known custom-endpoint mechanism at all. Not offered. |
|
||||
|
||||
Which harnesses show up in the Run-menu picker is read live off Codeman's own CLI registry,
|
||||
not a fixed list here, so this table can go stale before this page does — a greyed-out or
|
||||
missing entry is the more current answer.
|
||||
|
||||
## What it does not do
|
||||
|
||||
- **No remote or Docker sessions yet.** Both restart their agent differently under the hood
|
||||
(reattaching a durable tmux session rather than relaunching the process), so redirecting
|
||||
them needs its own plumbing that hasn't been built.
|
||||
- **No live hot-swap mid-conversation.** Applying a selection always restarts the process.
|
||||
- **No button to un-point a session from the UI yet.** Clearing back to native cloud is an
|
||||
HTTP call (`POST .../custom-model {"clear": true}`) or deleting the session; the settings
|
||||
panel manages saved endpoints, not what a running session is currently pointed at.
|
||||
- **Nothing is shared with your real cloud credentials.** The endpoint's own key, if any,
|
||||
never touches your Anthropic/OpenAI/Google login — a custom endpoint is a separate,
|
||||
explicit choice per session.
|
||||
|
||||
## Security
|
||||
|
||||
An endpoint's base URL can't point at a link-local or cloud-metadata address (both at save
|
||||
time and against the address it actually resolves to), the same guard Web Tabs uses for
|
||||
saved dashboards. Endpoint records and any per-session config files a harness needs are
|
||||
written with owner-only permissions. See
|
||||
[custom-model-endpoints-plan.md](https://github.com/Ark0N/Codeman/blob/master/docs/custom-model-endpoints-plan.md)
|
||||
in the repository for the full design reasoning, including why this feature closed a
|
||||
pre-existing gap in how session environment overrides were guarded rather than opening a new
|
||||
one.
|
||||
@@ -4,7 +4,7 @@ Run a case inside its own container instead of directly on your host: for isolat
|
||||
reproducible toolchain, and for the ability to pick the whole environment up and move it to
|
||||
another machine.
|
||||
|
||||
A docker case is a **location overlay**, not a run mode. All seven run modes work inside a
|
||||
A docker case is a **location overlay**, not a run mode. All ten run modes work inside a
|
||||
container. See [Core Concepts](Core-Concepts).
|
||||
|
||||
## One-time setup: the base image
|
||||
@@ -26,7 +26,7 @@ A zero exit code proves the layers ran, not that the toolchain works. Verify:
|
||||
|
||||
```bash
|
||||
docker run --rm codeman/agent:base bash -lc \
|
||||
'for c in claude codex gemini opencode agy pi; do printf "%-9s " $c; $c --version 2>&1 | head -1; done'
|
||||
'for c in claude codex gemini opencode agy pi grok dsh omp; do printf "%-9s " $c; $c --version 2>&1 | head -1; done'
|
||||
```
|
||||
|
||||
The image is secret-free. Credentials are delivered at runtime, never baked in, so exports
|
||||
@@ -79,6 +79,25 @@ Exactly one long-lived container per case, shared by every session in it.
|
||||
conversation** from the bind-mounted transcript.
|
||||
- Deleting the case removes the container. The workspace on the host survives.
|
||||
|
||||
## Attaching to a container you already run
|
||||
|
||||
Tick **Attach to an existing container** on **Add Case → Docker** to link a case to a
|
||||
container that already exists instead of creating one. Codeman only `exec`s into it and
|
||||
never creates, starts, stops, restarts or removes it, so a container that is missing or
|
||||
stopped fails with a message rather than being fixed for you. Drift detection does not
|
||||
apply (the container carries no Codeman configuration label). The full-image export is
|
||||
refused, since it would `docker commit` someone else's container, and the workspace export
|
||||
skips the pause that keeps an owned container consistent during the capture.
|
||||
|
||||
One adopted container can back several cases at different in-container directories, and
|
||||
**copy an existing case** pre-fills the form from a sibling on the same container. An exact
|
||||
twin (the same container and the same directory) is refused, as is a container another
|
||||
user adopted.
|
||||
|
||||
Adoption is **admin-only in multi-user mode**. Linking creates Codeman's own container
|
||||
with one bind mount that has already been checked; an adopted container's mounts belong to
|
||||
whoever started it, and one that mounts `/` hands the adopter the host.
|
||||
|
||||
## Credentials
|
||||
|
||||
Your existing host logins work inside the container without logging in again. Credentials
|
||||
@@ -92,10 +111,12 @@ the container instead.
|
||||
|
||||
Bind mounts are excluded from image capture, so exports stay secret-free.
|
||||
|
||||
One consequence worth knowing: Pi's credentials are seeded per file rather than as a whole
|
||||
directory, because that directory also holds sessions, extensions, and installed packages,
|
||||
which can be gigabytes. So in-container Pi sessions are invisible from the host, and `pi -c`
|
||||
inside a docker case sees only that container's history.
|
||||
One consequence worth knowing: Pi, Grok and OMP credentials are seeded per file rather than
|
||||
as whole directories, because those directories also hold sessions, extensions, downloads and
|
||||
installed packages, which can be gigabytes. So in-container Pi and Grok sessions are
|
||||
invisible from the host (`pi -c` and `grok -c` inside a docker case see only that
|
||||
container's history). OMP's `sessions/` is the exception and is shared read-write, because
|
||||
Codeman reads it host-side for history and resume.
|
||||
|
||||
## Isolation
|
||||
|
||||
|
||||
@@ -54,7 +54,10 @@ create-time sweep would yank the skill out from under other live sessions sharin
|
||||
directory. Remove them per case with `codeman skill uninstall --case <name>`.
|
||||
|
||||
The skill ships with the verb index always loaded, plus on-demand references for the verbs,
|
||||
worked multi-worker recipes, endpoint tables, and cross-session messaging.
|
||||
worked multi-worker recipes, endpoint tables, and cross-session messaging. It drives
|
||||
DeepSeek Harness workers the same way it drives Claude ones (`spawn_workers alpha
|
||||
beta:deepseek` is a mixed fleet in one call), since those are the two modes with real
|
||||
completion signals.
|
||||
|
||||
## The manual path
|
||||
|
||||
@@ -92,8 +95,9 @@ Read these before writing any code. Each one has cost somebody an afternoon.
|
||||
5. **Wait instead of polling, and a timeout is not an error.** The wait endpoints answer
|
||||
`200` with `wait.timedOut: true`. Loop over short waits rather than one long call, because
|
||||
tunnels cut idle connections.
|
||||
6. **Only `claude` sessions emit `stop` and `blocked`.** They come from Claude Code hooks.
|
||||
Shell and the external CLIs accept only `idle`, `working`, and `exit`; asking for `stop`
|
||||
6. **Only `claude` and `deepseek` sessions emit `stop` and `blocked`.** Claude's come from
|
||||
Claude Code hooks, DeepSeek's from the harness reporting its state to Codeman. Shell and
|
||||
the other external CLIs accept only `idle`, `working`, and `exit`; asking for `stop`
|
||||
explicitly there is a `400`, while omitting `until` is always safe. On a shell session
|
||||
`idle` fires **once at startup and never again**, so synchronize hook-less sessions with an
|
||||
output marker instead.
|
||||
@@ -130,7 +134,10 @@ curl -s -X POST "$API/api/sessions/$ID/input" \
|
||||
# Or wait for a marker in the output, which works on shell sessions too
|
||||
curl -s "$API/api/sessions/$ID/wait-output?contains=DONE_17909&from=buffer" | jq
|
||||
|
||||
# Read the terminal back
|
||||
# Read the last answer as clean text (claude, codex, deepseek sessions)
|
||||
curl -s "$API/api/sessions/$ID/last-response" | jq -r '.data.text'
|
||||
|
||||
# Or read the terminal back
|
||||
curl -s "$API/api/sessions/$ID/terminal?tail=4000" | jq -r '.data.output'
|
||||
|
||||
# Clean up, by exact id
|
||||
@@ -157,7 +164,13 @@ Make it unique per call, because tmux repaints replay old screen text.
|
||||
|
||||
### Reading output
|
||||
|
||||
Use `terminal?tail=`, not `/output`. The latter's text field is empty for every tmux-backed
|
||||
For `claude`, `codex` and `deepseek` sessions, read the answer from the transcript rather
|
||||
than the screen: `GET /api/sessions/:id/last-response` returns the last reply as clean text
|
||||
with no TUI frames or repaint noise. Poll it briefly rather than reading once, because the
|
||||
transcript lands slightly after the `stop` signal, so a read immediately after send-and-wait
|
||||
returns often comes back empty.
|
||||
|
||||
For everything else, use `terminal?tail=`, not `/output`. The latter's text field is empty for every tmux-backed
|
||||
session, which is every interactive session. `tail` counts **bytes**, and what comes back is
|
||||
terminal data with ANSI sequences included.
|
||||
|
||||
|
||||
@@ -21,6 +21,12 @@ No. Codeman drives agent CLIs you have already installed and logged in yourself.
|
||||
subscription or key that CLI uses is what pays for the tokens. Codeman never collects,
|
||||
stores, or refreshes your credentials.
|
||||
|
||||
### Which agent CLIs does it support?
|
||||
|
||||
Claude Code, OpenCode, Codex, Gemini, Antigravity, Pi, Grok Build, DeepSeek Harness and
|
||||
OMP, plus a plain shell, chosen per session. Claude is the reference mode and a few features
|
||||
are Claude-only; [Agent CLIs](Agent-CLIs) has the table.
|
||||
|
||||
### Does Codeman send my code or prompts anywhere?
|
||||
|
||||
No. There is no telemetry, no analytics, and no phone-home. The only network traffic
|
||||
|
||||
+18
-9
@@ -66,14 +66,14 @@ self-signed certificate, add `-k`.
|
||||
|
||||
## Endpoint map
|
||||
|
||||
Roughly 200 handlers across 24 route modules. By domain:
|
||||
Roughly 235 handlers across 26 route modules. By domain:
|
||||
|
||||
| Domain | Handlers | Covers |
|
||||
| ------------------- | -------- | --------------------------------------------------- |
|
||||
| System | 45 | Status, settings, search, digest, updates. |
|
||||
| Sessions | 34 | Create, input, terminal, wait, kill. |
|
||||
| Cases | 29 | Create, link, clone, remote and docker cases. |
|
||||
| Files | 16 | Preview, edit, raw, attachments, path picker. |
|
||||
| System | 56 | Status, settings, digest, updates, tunnel. |
|
||||
| Sessions | 34 | Create, input, terminal, wait, last response, kill. |
|
||||
| Cases | 34 | Create, link, clone, remote and docker cases. |
|
||||
| Files | 17 | Preview, edit, raw, attachments, path picker. |
|
||||
| Orchestrator | 10 | Plans and phases. |
|
||||
| Ralph | 9 | Loop control and configuration. |
|
||||
| Cron | 9 | Jobs and run history. |
|
||||
@@ -82,10 +82,12 @@ Roughly 200 handlers across 24 route modules. By domain:
|
||||
| Respawn | 7 | Respawn configuration and presets. |
|
||||
| Webviews | 6 | Saved dashboards, plus the proxy. |
|
||||
| Mux | 5 | tmux operations. |
|
||||
| Custom model endpoints | 5 | Saved OpenAI-compatible endpoints, and applying one to a session. |
|
||||
| Push | 4 | Web push subscriptions. |
|
||||
| Read My Mind | 4 | Intent profiles and prediction. |
|
||||
| Scheduled | 4 | The legacy scheduled-run concept. |
|
||||
| Approvals | 3 | The inbox and answering. |
|
||||
| Approvals | 4 | The inbox, answering, acknowledging. |
|
||||
| Tab layout | 2 | Named tab groups per owner. |
|
||||
| Teams, me, search, hooks, clipboard, telemetry, voice, ws | 1-2 each | |
|
||||
|
||||
Each route module documents its own endpoints in its file header.
|
||||
@@ -114,12 +116,13 @@ Three semantics that break callers who assume otherwise:
|
||||
`wait-output` matches a **literal substring, never a regex.** That is deliberate: no regex
|
||||
means no catastrophic backtracking on attacker-influenced output.
|
||||
|
||||
Only `claude` sessions emit `stop` and `blocked`, because those come from Claude Code hooks.
|
||||
Shell and external CLI sessions accept `idle`, `working`, and `exit`.
|
||||
Only `claude` and `deepseek` sessions emit `stop` and `blocked`: Claude's come from Claude
|
||||
Code hooks, DeepSeek's from the harness reporting its state to Codeman. Shell and the other
|
||||
external CLI sessions accept `idle`, `working`, and `exit`.
|
||||
|
||||
## SSE
|
||||
|
||||
`GET /api/events` is the live event stream. 156 event names, kept in sync between server and
|
||||
`GET /api/events` is the live event stream. 158 event names, kept in sync between server and
|
||||
client with a test that fails on drift.
|
||||
|
||||
The heartbeat is a **named** `sse:heartbeat` event rather than an SSE comment, because
|
||||
@@ -141,6 +144,12 @@ curl -s "$API/api/sessions" | jq '.data[].name' # live sessions
|
||||
curl -s "$API/api/sessions/unified" | jq # live + historical, deduped
|
||||
curl -s "$API/api/subagents" | jq # background agents
|
||||
curl -s "$API/api/search?q=deploy" | jq # cross-session search
|
||||
|
||||
# with ID set to a session id:
|
||||
curl -s "$API/api/sessions/$ID/last-response" | jq -r '.data.text' # last answer, from the transcript (claude, codex, deepseek)
|
||||
curl -s "$API/api/model-endpoints" | jq # saved custom OpenAI-compatible endpoints
|
||||
curl -s -X POST "$API/api/sessions/$ID/custom-model" -H 'Content-Type: application/json' \
|
||||
-d '{"endpointId":"local-llama","modelId":"qwen3-27b"}' | jq # restart the CLI on that endpoint; {"clear":true} undoes it
|
||||
```
|
||||
|
||||
## Limits
|
||||
|
||||
+4
-4
@@ -5,8 +5,8 @@
|
||||
<h3 align="center">Mission control for AI coding agents</h3>
|
||||
|
||||
Codeman runs your coding agents on your own machine and puts them behind one dashboard you
|
||||
can open from any device. It spawns Claude Code, OpenCode, Codex, Antigravity, Gemini, or
|
||||
Pi inside persistent tmux sessions, streams the real terminal to the browser, and keeps
|
||||
can open from any device. It spawns Claude Code, OpenCode, Codex, Antigravity, Gemini, Pi,
|
||||
Grok, DeepSeek Harness, or OMP inside persistent tmux sessions, streams the real terminal to the browser, and keeps
|
||||
working while you are away from the keyboard: it re-prompts idle agents, resumes when a
|
||||
subscription limit resets, runs jobs on a schedule, and shows every background subagent
|
||||
live.
|
||||
@@ -33,7 +33,7 @@ codeman web # then open http://localhost:3000
|
||||
|
||||
**Already running it**
|
||||
|
||||
- [Agent CLIs](Agent-CLIs) - the seven run modes, their setup, and which features are Claude-only.
|
||||
- [Agent CLIs](Agent-CLIs) - the ten run modes, their setup, and which features are Claude-only.
|
||||
- [Mobile Guide](Mobile-Guide) - phone and tablet use, QR login, the touch keyboard bar.
|
||||
- [Remote Access](Remote-Access) - Tailscale, Cloudflare tunnel, LAN plus password, QR login.
|
||||
- [Keeping Agents Running](Keeping-Agents-Running) - idle detection, respawn cycling, auto-resume on usage limits.
|
||||
@@ -122,7 +122,7 @@ codeman web # then open http://localhost:3000
|
||||
| OS | macOS or Linux. Windows works through WSL2. |
|
||||
| Node.js | 22 or newer. |
|
||||
| tmux | Required. Sessions live in tmux, which is what makes them survive restarts. |
|
||||
| An agent CLI | At least one of Claude Code, OpenCode, Codex, Gemini, Antigravity, Pi. Plain shell sessions need none. |
|
||||
| An agent CLI | At least one of Claude Code, OpenCode, Codex, Gemini, Antigravity, Pi, Grok Build, DeepSeek Harness, OMP. Plain shell sessions need none. |
|
||||
| Network | Binds to `127.0.0.1` by default. Reaching it from another device is a deliberate step: see [Remote Access](Remote-Access). |
|
||||
|
||||
Codeman is MIT licensed, self-hosted, and sends no telemetry. Everything runs on your
|
||||
|
||||
@@ -19,9 +19,11 @@ terminal into something that can notify you.
|
||||
| `teammate_idle` | An agent-team member goes idle. | Team surfaces. |
|
||||
| `task_completed` | A task finishes. | Task tracking, run summary. |
|
||||
|
||||
This is why several Codeman features are Claude-only. The other CLIs have no hook system, so
|
||||
for them Codeman watches terminal output, which reveals that something happened but not what
|
||||
it was.
|
||||
This is why several Codeman features are Claude-only. The one partial exception is DeepSeek
|
||||
Harness, whose terminal front door reports idle, working and blocked to Codeman over the
|
||||
harness's own supervisor contract, so it gets the hook-driven surfaces without any hook
|
||||
file. The other CLIs have no equivalent, so for them Codeman watches terminal output, which
|
||||
reveals that something happened but not what it was.
|
||||
|
||||
### How hooks get installed
|
||||
|
||||
@@ -67,7 +69,7 @@ sit beside the agents with no code at all. See [Web Tabs](Web-Tabs).
|
||||
### 2. SSE events
|
||||
|
||||
`GET /api/events` streams everything Codeman knows: session lifecycle, output, agent
|
||||
activity, approvals, cron runs. 155 named events, stable under semantic versioning.
|
||||
activity, approvals, cron runs. 158 named events, stable under semantic versioning.
|
||||
|
||||
This is the seam for anything that reacts. A bot that pings your chat channel when an agent
|
||||
needs a human is a short script over this stream.
|
||||
|
||||
@@ -27,6 +27,15 @@ The result is the property you want on a phone: a connection that drops mid-prom
|
||||
loses the prompt and never delivers it twice. Two browser tabs on the same session coexist,
|
||||
and only a reconnect from the *same* tab supersedes the old connection.
|
||||
|
||||
## Selecting and copying
|
||||
|
||||
Agent CLIs hold the mouse: clicks and drags are reported into the transcript rather than
|
||||
selecting text. `Shift+drag` starts a selection anyway, right-click copies it (with nothing
|
||||
selected the native context menu is left alone), and `Ctrl+Shift+C` copies without ever
|
||||
interrupting. **Auto Copy Selection** in **App Settings → Terminal & Input**, off by
|
||||
default, copies the moment you release the mouse. On phones, long-press selects; see
|
||||
[Mobile Guide](Mobile-Guide).
|
||||
|
||||
## Zero-lag local echo
|
||||
|
||||
On touch devices, keystrokes are painted in the terminal immediately and sent when you press
|
||||
@@ -55,6 +64,8 @@ reconcile against the real buffer and only apply while the cursor is on the comp
|
||||
Chinese, Japanese, and Korean input needs an IME, and an IME needs a real text field.
|
||||
Turning on CJK input in **App Settings → Terminal & Input** puts an always-visible textarea
|
||||
below the terminal that owns composition, then delivers the composed text to the session.
|
||||
Ctrl- and Alt-modified navigation keys typed through it reach the CLI as the modified
|
||||
sequences, so word jumps and history keys keep working.
|
||||
|
||||
## Voice dictation
|
||||
|
||||
|
||||
+70
-14
@@ -9,7 +9,7 @@ Getting Codeman onto a machine, verifying it works, updating it, and removing it
|
||||
| **macOS or Linux** | Windows works through WSL2. See [Windows](#windows-wsl) below. |
|
||||
| **Node.js 22+** | The installer offers to install it if missing. |
|
||||
| **tmux** | Not optional. Sessions live inside tmux, which is what makes them survive a server restart, a dropped connection, or a closed laptop. |
|
||||
| **An agent CLI** | At least one of [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), [Gemini CLI](https://github.com/google-gemini/gemini-cli), [Pi](https://pi.dev). Plain shell sessions need none. See [Agent CLIs](Agent-CLIs). |
|
||||
| **An agent CLI** | At least one of [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), [Gemini CLI](https://github.com/google-gemini/gemini-cli), [Pi](https://pi.dev), [Grok Build](https://github.com/xai-org/grok-build), [DeepSeek Harness](https://github.com/deepseek-ai/deepseek-harness), [OMP](https://github.com/can1357/oh-my-pi). Plain shell sessions need none. See [Agent CLIs](Agent-CLIs). |
|
||||
|
||||
Codeman itself sends no telemetry and phones no home. The only network traffic is your
|
||||
browser to your server, and whatever the agent CLI you chose does on its own.
|
||||
@@ -20,17 +20,26 @@ browser to your server, and whatever the agent CLI you chose does on its own.
|
||||
curl -fsSL https://getcodeman.com/install | bash
|
||||
```
|
||||
|
||||
This installs Node.js and tmux if they are missing, clones Codeman into `~/.codeman/app`,
|
||||
and builds it.
|
||||
This installs Node.js, tmux and a build toolchain if they are missing (node-pty ships no
|
||||
Linux prebuild, so it compiles from source), clones Codeman into `~/.codeman/app`, and
|
||||
builds it.
|
||||
|
||||
What it asks you:
|
||||
It starts by printing what it found (git, Node, tmux, build tools, agent CLIs, Tailscale,
|
||||
an existing install), then asks everything it needs up front, then does the work
|
||||
unattended. You can leave while it builds. What it asks you:
|
||||
|
||||
1. **Permission for every system change.** Package installs and agent CLI downloads are
|
||||
prompted individually. Nothing is installed silently.
|
||||
1. **One consent for the missing packages.** Git, Node.js, tmux and (on Linux) the build
|
||||
toolchain are installed after a single yes, and sudo asks for your password once for
|
||||
the whole run. Nothing is installed silently. If no agent CLI is found, a menu offers
|
||||
to install any of them (DeepSeek excepted: its npm package installs only a launcher
|
||||
with no runnable profile), or you skip and install one yourself later.
|
||||
2. **How the dashboard should be reachable.** Three choices:
|
||||
- **Tailscale** (recommended for phone access): keeps the loopback bind and walks you
|
||||
through `tailscale serve`, including the tailnet HTTPS toggle, then verifies the result
|
||||
end to end.
|
||||
- **Tailscale** (recommended for phone access): keeps the loopback bind, installs
|
||||
Tailscale if needed, logs in, enables the tailnet HTTPS toggle (it opens the admin
|
||||
page for you and waits; Ctrl+C there skips Tailscale for this run), then configures `tailscale serve` after the build and
|
||||
verifies the result end to end. If another app already owns `:443` on your node,
|
||||
you choose between a sub-path (`https://<machine>.<tailnet>.ts.net/codeman`, the
|
||||
default), a second port, replacing the other mapping, or skipping.
|
||||
- **Your local network** (`0.0.0.0`): prompts for a password. Skipping the password takes
|
||||
an explicit confirmation and ends on a loud warning.
|
||||
- **This machine only** (`127.0.0.1`): the safest option, and the default for a bare
|
||||
@@ -41,26 +50,54 @@ What it asks you:
|
||||
Tailscale. An existing loopback install defaults to keeping loopback, or to Tailscale when
|
||||
a serve mapping for Codeman is already there. A bare Enter never pulls in new software,
|
||||
and a non-interactive run always keeps the safe loopback default.
|
||||
3. **What to do when it finishes.** Run in this terminal, install as a background service
|
||||
that starts on boot, or do nothing yet.
|
||||
3. **What to call this machine on your tailnet** (Tailscale route only). By default the URL
|
||||
uses the machine's existing name. Answer yes to rename it `codeman-<hostname>`; the
|
||||
default is no, because the tailnet name is also what SSH and everything else on that
|
||||
machine are reached by.
|
||||
4. **Whether to run Codeman in the background.** Enter installs a systemd user service or a
|
||||
macOS LaunchAgent that starts on boot; answering no offers to start it in this terminal
|
||||
instead, or not at all.
|
||||
|
||||
It ends on a screen with the URL (your tailnet, your network, or this machine), a QR code to
|
||||
scan with your phone, and the two commands you need to manage the service.
|
||||
|
||||
Re-running the same one-liner **updates an existing install in place**. Local changes in
|
||||
`~/.codeman/app` are stashed rather than discarded, a running service is restarted and
|
||||
verified, and your existing network binding is preserved. An interrupted first install
|
||||
resumes instead of restarting.
|
||||
|
||||
Two other entry points exist:
|
||||
Other entry points:
|
||||
|
||||
```bash
|
||||
install.sh status # print the URLs, the QR code and the manage commands again
|
||||
install.sh update # update only
|
||||
install.sh uninstall # remove
|
||||
install.sh uninstall # remove (offers to undo a rename it performed)
|
||||
install.sh tailscale # retrofit Tailscale access onto an existing install
|
||||
install.sh name [<n>] # rename this machine on your tailnet (default codeman-<hostname>)
|
||||
install.sh cloudflared # install cloudflared for the in-app Cloudflare tunnel
|
||||
```
|
||||
|
||||
**Flags** answer the questions from the command line and pipe through `bash -s --`:
|
||||
|
||||
```bash
|
||||
curl -fsSL https://getcodeman.com/install | bash -s -- --tailscale --service
|
||||
curl -fsSL https://getcodeman.com/install | bash -s -- --lan --password 'x' --service
|
||||
curl -fsSL https://getcodeman.com/install | bash -s -- --local --run
|
||||
```
|
||||
|
||||
`--tailscale` / `--lan` / `--local` answer the access question, `--name <n>` / `--no-rename`
|
||||
the name, `--service` / `--run` / `--no-start` the last one. `--yes` takes every default
|
||||
(it still waits on a Tailscale login URL, and a network bind still asks for a password).
|
||||
`--port <n>` moves Codeman off 3000; the service file and the serve mapping follow it. On an
|
||||
existing install, `--port` and `--password` re-run the setup so the service file picks them up,
|
||||
and a re-run with `--lan` or `--tailscale` keeps the password the service already has.
|
||||
|
||||
**Automation and CI**: with no terminal attached, any step that would change the system
|
||||
aborts with instructions instead of running silently. Set `CODEMAN_NONINTERACTIVE=1` to
|
||||
approve those steps. `CODEMAN_TAILSCALE=1` preselects the Tailscale answer, and never
|
||||
installs Tailscale itself non-interactively.
|
||||
installs Tailscale itself non-interactively; a non-interactive run never renames the
|
||||
machine and never starts a service. Everything the unattended steps print goes to
|
||||
`~/.codeman/install.log`, and the last lines of it are shown when a step fails.
|
||||
|
||||
## Route B: npm
|
||||
|
||||
@@ -101,6 +138,21 @@ at server start, so markup changes need a restart.
|
||||
|
||||
See [Contributing](Contributing) for the rest of the development loop.
|
||||
|
||||
## Route D: Docker Compose
|
||||
|
||||
Codeman itself can run in a container and spawn Docker cases as sibling containers through
|
||||
the host's Docker socket. Copy `docker/.env.example` to `docker/.env`, set
|
||||
`CODEMAN_PASSWORD`, then:
|
||||
|
||||
```bash
|
||||
bash docker/Start-Codeman.sh
|
||||
```
|
||||
|
||||
Run the script again after updating rather than a plain `docker compose up`, so the rebuilt
|
||||
image, the refreshed volumes and the entrypoint arrive together. The full guide, including
|
||||
storage and networking options, is
|
||||
[`docker/README.md`](https://github.com/Ark0N/Codeman/blob/master/docker/README.md).
|
||||
|
||||
## Installing an agent CLI
|
||||
|
||||
Codeman drives CLIs, it does not bundle them. Install at least one:
|
||||
@@ -113,6 +165,9 @@ Codeman drives CLIs, it does not bundle them. Install at least one:
|
||||
| **Antigravity** | See [antigravity.google](https://antigravity.google) | Google's successor to the consumer Gemini CLI. |
|
||||
| **Gemini CLI** | See [github.com/google-gemini/gemini-cli](https://github.com/google-gemini/gemini-cli) | Enterprise only since Google's June 2026 consumer cutover. |
|
||||
| **Pi** | See [pi.dev](https://pi.dev) | No permission prompts and no sandbox by design. Read [Agent CLIs](Agent-CLIs) before using it on a repo you care about. |
|
||||
| **Grok Build** | `curl -fsSL https://x.ai/cli/install.sh \| bash` | xAI. Lands in `~/.grok/bin`; `grok login --device-auth` for headless hosts. |
|
||||
| **DeepSeek Harness** | `npm i -g @deepseek-ai/dsh pnpm`, then a terminal profile | The npm package is only a launcher. Codeman's Run menu installs the community terminal profile for you. See [Agent CLIs](Agent-CLIs). |
|
||||
| **OMP** | `curl -fsSL https://omp.sh/install \| sh` | Oh My Pi. Run it once by hand to finish its own onboarding. |
|
||||
|
||||
Log each CLI in once, by hand, before pointing Codeman at it. Codeman never collects or
|
||||
stores your CLI credentials.
|
||||
@@ -161,6 +216,7 @@ Full detail, including logs and the self-updater, is in
|
||||
| Installer | Re-run the one-liner, or **App Settings → System → Updates** in the UI. |
|
||||
| npm | `npm update -g aicodeman` |
|
||||
| git clone | `git pull && npm install && npm run build`, then restart. |
|
||||
| Docker Compose | Re-run `Start-Codeman.sh`. The in-app updater works too, and refuses a release that changes the container definition until you re-run the script. |
|
||||
|
||||
The in-app updater covers git-clone installs supervised by systemd or launchd. It restarts
|
||||
the process that is running it, so the actual work happens in a detached script and the
|
||||
|
||||
@@ -27,9 +27,12 @@ keystroke echo. Idle now lands a few seconds after a turn genuinely ends.
|
||||
There are several layers stacked on that: a completion message from the CLI, an AI check,
|
||||
output silence, and token stability.
|
||||
|
||||
**For every other CLI**, there are no hooks to lean on, so detection is output
|
||||
stabilization: the session is idle when output stops changing. Coarser, and it is why the
|
||||
features further down this page are Claude-only.
|
||||
**For the other CLIs** it depends on what the CLI tells Codeman. Codex declares its own
|
||||
prompt glyph and working line, so it gets the same screen check Claude does (before 1.26.1
|
||||
every Codex session reported idle for its whole life). DeepSeek Harness reports idle,
|
||||
working and blocked to Codeman itself, which is as precise as hooks. Everything else is
|
||||
output stabilization: the session is idle when output stops changing. Coarser, and it is
|
||||
why the features further down this page are Claude-only.
|
||||
|
||||
## The Respawn Controller
|
||||
|
||||
@@ -101,13 +104,18 @@ subscription plan.
|
||||
**Claude only.** A header chip showing live subscription usage, on by default on desktop and
|
||||
off on phones.
|
||||
|
||||
It works by installing a status line exporter into Claude Code, which posts Claude's own
|
||||
rate limit data back to Codeman. The exporter is marker-identified, so it only ever touches
|
||||
a status line Codeman installed, never one you wrote yourself, and it prints your footer
|
||||
through so the in-terminal status line still works.
|
||||
It works through a status line exporter that Codeman hands to `claude` as an ephemeral
|
||||
setting when it spawns the session, never written to disk, which posts Claude's own rate
|
||||
limit data back to Codeman. Your own status line (project-local, project, then
|
||||
`~/.claude/settings.json`) is wrapped and printed through, and a `claude` you run by hand
|
||||
outside Codeman sees nothing of it. Workspaces an older Codeman wrote the exporter into are
|
||||
cleaned up the first time a session starts there. Codex limits come from a read-only poll of
|
||||
its own app-server. Known limit: sessions inside a Docker case do not feed the chip yet.
|
||||
|
||||
The chip and the exporter are the same setting. Turning the chip on without the exporter
|
||||
would leave it showing a dash forever, so resolve it in one place: **App Settings**.
|
||||
would leave it showing a dash forever, so resolve it in one place: **App Settings**. A
|
||||
device writes the switch only when it flips the chip, so a phone (chip off by default)
|
||||
saving its font size cannot switch collection off for your desktop.
|
||||
|
||||
## Circuit breakers
|
||||
|
||||
|
||||
@@ -30,6 +30,11 @@ Press `Ctrl+?` in the app for the same list in a floating overlay.
|
||||
| `Ctrl+Shift+R` | Restore terminal size. |
|
||||
| `Ctrl` `+` / `Ctrl` `-` | Font size. |
|
||||
| `Shift+Wheel` | Scroll the local buffer, even where the wheel is forwarded to the CLI. |
|
||||
| `Shift+drag` | Start a selection in a pane whose mouse events go to the CLI. |
|
||||
| Right-click | Copy the selection. With nothing selected the native menu is left alone. |
|
||||
| `Ctrl+Z` | Swallowed in agent sessions so a running CLI cannot be suspended. Normal job control in a shell. |
|
||||
|
||||
Anything you copy is cleaned on the way to the clipboard: each line loses the padding spaces a full-screen program paints across the rest of the row. Leading indentation is left exactly as it is, so indented code, a `git log` message body and `git diff` context lines paste back the way they looked on screen. An `Alt+drag` rectangular selection is copied exactly as it looks, so its columns stay lined up.
|
||||
|
||||
## Everything else
|
||||
|
||||
|
||||
@@ -30,8 +30,12 @@ require a secure context.
|
||||
| Toolbar | Bottom: Run, Stop, **Enter**, case picker, voice, settings. |
|
||||
| Keyboard bar | Above the on-screen keyboard when it is open. |
|
||||
|
||||
Layout respects notch and home-indicator safe areas, touch targets are 44px, and the case
|
||||
picker is a bottom sheet rather than a dropdown.
|
||||
The phone layout applies up to 599px of viewport width, so the Plus and Pro Max iPhones,
|
||||
the Pixel Pro and a folded Z Fold get it too; wider devices get the tablet layout. Layout
|
||||
respects notch and home-indicator safe areas, touch targets are 44px, and the case picker is
|
||||
a bottom sheet rather than a dropdown. On a folding phone (iPhone Duo) dialogs stay clear of
|
||||
the hinge, and opening or closing the device is treated as the device changing shape, never
|
||||
as the keyboard appearing.
|
||||
|
||||
**Swipe left and right** on the terminal to switch sessions.
|
||||
|
||||
@@ -58,7 +62,9 @@ A row of keys above the virtual keyboard, and what it contains depends on the se
|
||||
|
||||
**Agent sessions** get quick actions: `/init`, `/clear`, `/compact`, a clipboard key, `Esc`,
|
||||
a path picker, an image key, and 🧠 when Read My Mind is on. Destructive commands need a
|
||||
double press, so you cannot fire `/clear` with a stray thumb.
|
||||
double press, so you cannot fire `/clear` with a stray thumb. On Codex sessions the bar also
|
||||
shows `⇧←` and `⇧→`, the Shift-modified arrows Codex binds to editing the last queued
|
||||
message and walking the prompt stack.
|
||||
|
||||
**Shell sessions** automatically swap it for terminal controls: `Ctrl`, `Esc`, `Tab`, four
|
||||
arrows, paste, and dismiss. Your normal preference is remembered and restored when you
|
||||
|
||||
@@ -33,8 +33,8 @@ reloading the dashboard while a permission dialog is blocking a session does not
|
||||
with a normal-looking tab.
|
||||
|
||||
For Claude sessions, these come from Claude Code's hooks and are precise about *why* the
|
||||
session stopped. For other CLIs there are no hooks, so you get the coarser output-based
|
||||
signal.
|
||||
session stopped; DeepSeek Harness sessions report the same states themselves. For the other
|
||||
CLIs there are no hooks, so you get the coarser output-based signal.
|
||||
|
||||
## Window title and OS notifications
|
||||
|
||||
@@ -62,7 +62,8 @@ Once subscribed, a blocking prompt reaches your phone even from a locked screen.
|
||||
|
||||
## The Approvals Inbox
|
||||
|
||||
**Opt-in, off by default. Claude sessions only.**
|
||||
**Opt-in, off by default. Claude sessions, plus DeepSeek Harness sessions, whose terminal
|
||||
front door reports its prompts to Codeman.**
|
||||
|
||||
One queue of every prompt currently waiting on a human, across all your sessions, answerable
|
||||
in place. When you have eight workers running, this is the difference between checking eight
|
||||
@@ -136,7 +137,8 @@ from the lock screen.
|
||||
- **No push over plain HTTP.** It is a browser requirement, not a Codeman one.
|
||||
- **iOS needs the home screen install.** A Safari tab will never receive push.
|
||||
- **The bell is invisible at zero.** That is deliberate, not a broken setting.
|
||||
- **Approvals are Claude-only.** They are built on hook events the other CLIs do not emit.
|
||||
- **Approvals need real signals.** They are built on hook events, which Claude emits and
|
||||
DeepSeek Harness reports itself; the other CLIs do neither.
|
||||
- **A stale menu answer is refused, not sent.** If you answer a card for a dialog that has
|
||||
since gone away, Codeman declines rather than typing a digit into the composer.
|
||||
|
||||
|
||||
@@ -67,6 +67,9 @@ one:
|
||||
| **Gemini** | Enterprise only since Google's consumer cutover. |
|
||||
| **Antigravity** | Google's successor to the consumer Gemini CLI. |
|
||||
| **Pi** | No permission prompts and no sandbox by design. |
|
||||
| **Grok Build** | xAI's CLI. |
|
||||
| **DeepSeek Harness** | Needs a terminal profile; the menu offers to install one. |
|
||||
| **OMP** | Oh My Pi, configured entirely through its own `~/.omp`. |
|
||||
| **Terminal / Shell** | A plain shell, no agent. Also the **Run Shell** button. |
|
||||
|
||||
The dropdown also lists any saved dashboard URLs ([Web Tabs](Web-Tabs)) and your recent
|
||||
|
||||
@@ -33,11 +33,12 @@ Your devices join a private network, and Codeman stays bound to loopback. Nothin
|
||||
published to the internet, and you get real HTTPS with a real certificate.
|
||||
|
||||
The installer sets this up for you, including installing Tailscale, logging in, enabling
|
||||
tailnet HTTPS, and verifying the result end to end. To retrofit it onto an existing
|
||||
install:
|
||||
tailnet HTTPS, and verifying the result end to end. It ends on the URL with a QR code to
|
||||
scan. To retrofit it onto an existing install, or to see the URL and QR code again:
|
||||
|
||||
```bash
|
||||
install.sh tailscale
|
||||
install.sh status
|
||||
```
|
||||
|
||||
By hand:
|
||||
@@ -49,6 +50,22 @@ tailscale serve status
|
||||
|
||||
Then open `https://<machine>.<tailnet>.ts.net` from any device on your tailnet.
|
||||
|
||||
### The name in the URL
|
||||
|
||||
The URL is the machine's MagicDNS name, so on a machine called `tnode` it is
|
||||
`https://tnode.<tailnet>.ts.net`. Three ways to influence that, from least to most work:
|
||||
|
||||
| You want | How |
|
||||
| ------------------------------------------ | ----------------------------------------------------------------------------------------------------- |
|
||||
| The machine's existing name (default) | Nothing. This is what the installer does unless you say otherwise. |
|
||||
| `https://codeman-<hostname>.<tailnet>.ts.net` | Answer yes to the installer's name question, pass `--name codeman-<hostname>`, or run `install.sh name`. This renames the machine tailnet-wide (SSH included), which is why the installer defaults to no. `install.sh uninstall` offers to rename it back. |
|
||||
| `https://codeman.<tailnet>.ts.net` | A [Tailscale Service](https://tailscale.com/docs/features/tailscale-services). Only a **tagged** node can host one (a device signed in with a user account cannot), the service is defined and approved in the admin console, and the feature is in beta. The installer does not set this up; it is a `tailscale serve --service=svc:codeman --https=443 127.0.0.1:3000` on a tagged host once the service exists. |
|
||||
|
||||
If `:443` on your node already belongs to another app, the installer offers Codeman under
|
||||
`https://<machine>.<tailnet>.ts.net/codeman` (the default, via `tailscale serve --set-path`
|
||||
plus Codeman's `--base-url`), on a second port (`https://<machine>.<tailnet>.ts.net:8443`),
|
||||
or replacing the other mapping. It never replaces anything without asking.
|
||||
|
||||
Notes:
|
||||
|
||||
- Keep the loopback bind. `tailscale serve` connects to `127.0.0.1:3000` locally, so
|
||||
@@ -58,7 +75,11 @@ Notes:
|
||||
- Codeman's Host-header allowlist already accepts `.ts.net`, so no extra configuration is
|
||||
needed.
|
||||
- The installer never resets or rewrites `serve` mappings other than the one pointing at
|
||||
Codeman's port, so unrelated serve configuration is left alone.
|
||||
Codeman's port, so unrelated serve configuration is left alone. It also never opens a
|
||||
`tailscale funnel` (that is the public internet) and never advertises a Tailscale Service.
|
||||
- On macOS, the App Store and standalone Tailscale apps only run once someone is logged in,
|
||||
so a headless Mac needs the open-source `tailscaled` for the URL to come back after a
|
||||
reboot on its own.
|
||||
|
||||
## Cloudflare tunnel
|
||||
|
||||
|
||||
@@ -4,7 +4,7 @@ Point a case at another machine and the agent runs **there**, with the same dash
|
||||
mobile UI, and autonomy features. Your laptop becomes a window onto a session living on the
|
||||
remote host.
|
||||
|
||||
Like Docker, this is a **location overlay** on a case, not a run mode. All seven run modes
|
||||
Like Docker, this is a **location overlay** on a case, not a run mode. All ten run modes
|
||||
work remotely. See [Core Concepts](Core-Concepts).
|
||||
|
||||
## Why bother
|
||||
@@ -52,7 +52,10 @@ A watcher with bounded backoff notices a dead SSH pane and quietly reattaches to
|
||||
running remote session. On by default; the kill switch is in
|
||||
**App Settings → Agents & CLIs → Remote auto-reconnect**.
|
||||
|
||||
Intentional kills are never revived. Closing a session means closing it.
|
||||
Intentional kills are never revived. Closing a session means closing it. Neither is a clean
|
||||
exit inside the pane (Ctrl-D, `exit`, Ctrl-C at the CLI's prompt): that tears the remote
|
||||
tmux session down, and the watcher revives a session only when that durable session is
|
||||
verifiably still alive. Only a transport drop is reconnected.
|
||||
|
||||
## Discover and attach
|
||||
|
||||
@@ -70,6 +73,14 @@ Attaching to someone else's session and closing your tab must not end their run,
|
||||
not. Several clients can attach the same remote session at different window sizes without
|
||||
clamping each other, and discovery shows a shared badge with the client count.
|
||||
|
||||
## Files
|
||||
|
||||
Previews, downloads and text reads in a remote case go over the same ssh connection the
|
||||
session uses, so a clicked path opens the file on the machine the agent is on, `Range`
|
||||
seeking included. Nothing is copied to the Codeman host. Editing, Office previews,
|
||||
thumbnails, the file tree and the tail viewer are not available remotely and answer a clear
|
||||
400 rather than a misleading 404. Details in [Working With Files](Working-With-Files).
|
||||
|
||||
## Security
|
||||
|
||||
Every SSH command line in Codeman flows through one builder that shell-escapes every
|
||||
|
||||
@@ -67,6 +67,14 @@ On Linux, if you want the service running while you are not logged in:
|
||||
loginctl enable-linger $USER
|
||||
```
|
||||
|
||||
On macOS, a LaunchAgent starts when you log in, not at boot. A headless Mac (no GUI login)
|
||||
needs a system LaunchDaemon instead, written by hand as root. The installer recognises an
|
||||
existing `/Library/LaunchDaemons/com.codeman.web.plist` and leaves it alone rather than
|
||||
installing a LaunchAgent next to it, since the two would fight over the port; remove the
|
||||
daemon first if you want to switch. The same login caveat applies to the App Store and
|
||||
standalone Tailscale apps, so on a headless Mac the Tailscale URL only comes back after a
|
||||
reboot if the open-source `tailscaled` is used.
|
||||
|
||||
### Writing the unit by hand
|
||||
|
||||
**Linux (systemd user unit):**
|
||||
@@ -139,6 +147,7 @@ log stream --predicate 'process == "node"' # macOS, noisy
|
||||
| Installer | Re-run the one-liner, or **App Settings → System → Updates**. |
|
||||
| npm | `npm update -g aicodeman` |
|
||||
| git clone | `git pull && npm install && npm run build`, then restart. |
|
||||
| Docker Compose | Re-run `Start-Codeman.sh`, or the in-app updater, which restarts the container in place. |
|
||||
|
||||
### The in-app updater
|
||||
|
||||
@@ -170,6 +179,18 @@ service without colliding with the main one. `CODEMAN_DATA_DIR` and `CODEMAN_TMU
|
||||
exist for the rare case where they need to differ, but setting only one of them recreates
|
||||
exactly the problem you were avoiding.
|
||||
|
||||
## Running Codeman itself in Docker
|
||||
|
||||
The Compose deployment in `docker/` runs the server in a container and spawns Docker cases
|
||||
as sibling containers through the mounted host socket. Start it with
|
||||
`bash docker/Start-Codeman.sh` rather than a bare `docker compose up`: the script pre-creates
|
||||
the bind-mounted directories with the right owner, honours a `docker-compose.override.yml`,
|
||||
and refreshes the build volumes when the checkout moved under them. The in-app updater
|
||||
applies code only and restarts by letting the container exit, so it refuses a release that
|
||||
changes the Dockerfile, the compose file, or adds a new `.env` key, until you re-run the
|
||||
script. Guide:
|
||||
[`docker/README.md`](https://github.com/Ark0N/Codeman/blob/master/docker/README.md).
|
||||
|
||||
## The tunnel as a service
|
||||
|
||||
```bash
|
||||
|
||||
@@ -67,6 +67,7 @@ be wrong for at least one of them:
|
||||
| **File Viewer** | Real path resolution before boundary checks, so symlinks cannot escape. Sensitive trees blocked. Edit mode adds an extension allowlist, a size cap, `.git` denial, and optimistic concurrency. It never creates files. |
|
||||
| **Attachments** | An id-based registry, so browser requests never carry absolute paths. The magic-link scanner is prompt-injectable by nature and is therefore force-confined to the session's workspace. Extension allowlist, not a blocklist. |
|
||||
| **Path picker** | Its own root allowlist rather than the workspace confinement. In multi-user mode a non-admin gets only their own user space, because per-user spaces live inside the home directory. |
|
||||
| **Remote cases** | Reads go over the session's own ssh connection and are resolved and contained on the remote host, with a bounded number of ssh children. Nothing is copied to the Codeman host; writes, Office previews and thumbnails are refused. |
|
||||
|
||||
Downloads block sensitive paths outright (`.env`, credentials files, `~/.ssh`, AWS
|
||||
credentials), and SVG and HTML are served as downloads with `nosniff` so they cannot execute
|
||||
|
||||
@@ -46,6 +46,7 @@ supervised by systemd or launchd; npm installs report as non-updatable. See
|
||||
| Extended Keyboard Bar | Per device | Which accessory bar phones get. Shell sessions override it while they are active. |
|
||||
| Wheel Scrolls Local History | Off | Keeps the wheel on the local buffer instead of forwarding it to the CLI. |
|
||||
| Auto Copy Selection | Off | Copies highlighted terminal text to the clipboard the moment you finish selecting it. Ctrl+C still copies on demand. |
|
||||
| Normal / Bold font weight | xterm defaults | Per device, each slot from 100 to 900. The bundled JetBrains Mono renders every step, so a lighter normal weight makes Claude's bold headings stand out. Applies live to the terminal, both echo overlays and open team panes. |
|
||||
| WebGL Renderer | On | With a GPU-stall watchdog that falls back to DOM rendering. |
|
||||
| Gesture Control | Off | Camera hand tracking. Also needs `CODEMAN_GESTURE=1` on the server. |
|
||||
|
||||
@@ -72,10 +73,13 @@ every session or only the active tab.
|
||||
| Entrance Animations | Per-surface animation styles for tabs, terminals, windows, and lineage lines. All default to the legacy no-animation behaviour. |
|
||||
| Display Name | Your name in the UI. Cosmetic only; it never renames the package, CLI, API, or storage. |
|
||||
| Interface Language | English or Simplified Chinese. Per device. |
|
||||
| Session List Layout | Header tab strip (default) or a collapsible left sidebar. See [The Dashboard](The-Dashboard#session-list-layout). |
|
||||
| Session List Layout | Header tab strip (default), a collapsible left sidebar, or the sidebar with detailed rows. See [The Dashboard](The-Dashboard#session-list-layout). |
|
||||
| Tab Orientation | Keeps the header list but turns the strip vertical beside the terminal, resizable, with detailed rows by default. Desktop and tablet only. |
|
||||
| Vertical Rail Order | *By activity* (default) sorts the rail the way the home screens are sorted; *Manual* keeps your tab order and drag-reordering. |
|
||||
| Tall Tabs | Taller tab strip. |
|
||||
| Pop-out Button on Tabs | Adds the detach control to tabs, with a per-tab override. |
|
||||
| Spawn Lineage Lines | Arcs from a parent tab to sessions it spawned. Desktop only, on by default. |
|
||||
| Auto-name Sessions | Titles a new tab after its first prompt, keeping the case prefix (`w3-myapp: fix the login redirect`). Synced, off by default. See [The Dashboard](The-Dashboard#automatic-session-names). |
|
||||
| Overview Home Screen | The phone home screen. On by default. |
|
||||
|
||||
### Models
|
||||
@@ -88,6 +92,10 @@ Model and effort are both **soft defaults**: the model is written into the case'
|
||||
`.claude/settings.local.json` and effort is passed at start, so `/model` and `/effort`
|
||||
inside a session override them at any time.
|
||||
|
||||
**Custom model endpoints** (off by default) adds a saved-endpoint list plus a matching
|
||||
section to the Run dropdown, for pointing a harness at your own OpenAI-compatible server
|
||||
instead of its native cloud backend. See [Custom Model Endpoints](Custom-Model-Endpoints).
|
||||
|
||||
### Agents & CLIs
|
||||
|
||||
| Setting | Notes |
|
||||
@@ -155,6 +163,9 @@ Some things are configured before the server starts, not in the UI:
|
||||
| `CODEMAN_DOCKER_BRIDGE_HOOKS` | Lets in-container hooks reach the host on a loopback bind. |
|
||||
| `CODEMAN_FILE_PICKER_ROOTS` | Extra roots for the path picker. |
|
||||
| `CODEMAN_ALLOW_UNAUTHENTICATED_NETWORK` | Acknowledges exposing the server with no password. |
|
||||
| `CODEMAN_BASE_URL` | Mounts Codeman under a sub-path behind a reverse proxy that forwards the prefix unchanged. See [Remote Access](Remote-Access). |
|
||||
| `CODEMAN_MAX_DOWNLOAD_BYTES` | Cap on raw file bodies and downloads. 2 GB by default, `0` for none. |
|
||||
| `CODEMAN_MAX_REMOTE_FILE_SSH` | Concurrent ssh reads for files in remote cases. 4 by default. |
|
||||
|
||||
## Gotchas
|
||||
|
||||
|
||||
@@ -22,12 +22,14 @@ page says so and names the setting.
|
||||
|
||||
The session list lives in the header as a horizontal strip by default. With a lot of
|
||||
sessions open that strip stops being scannable, so **App Settings → Appearance → Tabs →
|
||||
Session List Layout** can move it into a vertical sidebar on the left instead.
|
||||
Session List Layout** can move it into a vertical sidebar on the left instead, and
|
||||
**Tab Orientation** can turn the strip itself into a vertical rail.
|
||||
|
||||
| Layout | Behaviour |
|
||||
| -------------------- | --------------------------------------------------------------------------------- |
|
||||
| **Header tab strip** | The default. Wraps to a second row on desktop, scrolls sideways on a phone. |
|
||||
| **Left sidebar** | A vertical list with a filter box and a live session count. `Alt+B` collapses it to a narrow rail that keeps the status dots and task badges visible. On a phone it is an off-canvas drawer rather than a docked rail. |
|
||||
| **Left sidebar** | A vertical list with a filter box and a live session count. `Alt+B` collapses it to a narrow rail that keeps the status dots and task badges visible. On a phone it is an off-canvas drawer rather than a docked rail. A detailed variant adds the home screen's per-session line (`created 3d ago · working 12m`) and a status pill. |
|
||||
| **Vertical rail** | The strip turned vertical beside the terminal, resizable, with detailed rows by default. **Vertical Rail Order** sorts it by activity (blocked on you first, then longest running, then most recently quiet), the same order as the home screens; pick *Manual* to get your own order and drag-reordering back. Desktop and tablet only. |
|
||||
|
||||
It is the same list either way, just re-hosted: tab order, drag-to-reorder, the `Alt+1`
|
||||
to `Alt+9` numbers and every status colour below behave identically in both. The setting is
|
||||
@@ -66,6 +68,18 @@ reloading while a permission prompt is blocking does not lose the red tab.
|
||||
|
||||
Tabs can also be dragged to reorder.
|
||||
|
||||
### Automatic session names
|
||||
|
||||
Off by default. Turn on **Auto-name Sessions** (App Settings → Appearance → Tabs; synced
|
||||
across devices) and a tab that still carries its generated name, such as `w3-myapp`, takes a
|
||||
title from the first real prompt you submit, keeping the prefix: `w3-myapp: fix the login
|
||||
redirect`. The strip shows the title and keeps the prefix in the tooltip, and the next
|
||||
session in that case still counts up to `w4-myapp`. It happens once per session, only for
|
||||
prompts you type or send through the input API (never a Ralph, respawn, cron or approval
|
||||
answer), and never for shells. Slash commands such as `/clear` do not become titles; the
|
||||
next prompt gets its turn. A name you set yourself, before or after, is never touched. The
|
||||
title is derived locally from the prompt's first sentence; no text leaves the machine.
|
||||
|
||||
On phones the strip scrolls horizontally instead of wrapping, and the active tab is always
|
||||
scrolled into view. It is not reordered to the front, so the `Alt+N` numbering stays stable.
|
||||
|
||||
@@ -142,6 +156,10 @@ Worth knowing:
|
||||
always local scrollback. Other CLIs scroll locally.
|
||||
- **Selection copy.** `Ctrl+C` copies when text is selected and interrupts when it is not.
|
||||
`Ctrl+Shift+C` always copies.
|
||||
- **Selecting where the CLI owns the mouse.** `Shift+drag` starts a selection even in a pane
|
||||
whose mouse events are forwarded to the CLI, and right-click copies the selection (with
|
||||
nothing selected the native menu is left alone). **Auto Copy Selection** in App Settings
|
||||
copies the moment you release.
|
||||
- **Zero-lag input.** On touch devices, keystrokes paint locally before the round trip. See
|
||||
[Input And Voice](Input-And-Voice).
|
||||
- **Renderer.** WebGL by default, with a watchdog that falls back to DOM rendering if the
|
||||
@@ -155,8 +173,9 @@ which lists past sessions including Claude conversations started outside Codeman
|
||||
|
||||
Two extras depending on the device:
|
||||
|
||||
- **Desktop, wide windows**: your open tabs appear as a rail docked to the left edge, in tab
|
||||
order, with created and last-active stamps. It needs at least 1180px of width; below that
|
||||
- **Desktop, wide windows**: your open tabs appear as a rail docked to the left edge, in
|
||||
overview order (blocked on you first, then longest running, then most recently quiet),
|
||||
with created and state-duration stamps. It needs at least 1180px of width; below that
|
||||
it is hidden so it cannot overlap the search panel.
|
||||
- **Phones**: tapping the "C" logo gives a session overview instead: NEEDS YOU first, then
|
||||
current sessions, then past ones. On by default.
|
||||
@@ -192,7 +211,9 @@ so it is fast and cannot be turned into a traversal.
|
||||
## Appearance
|
||||
|
||||
**App Settings → Appearance** carries the theme skins, including light ones. The choice is
|
||||
applied before the first paint, so there is no flash of the wrong theme on load.
|
||||
applied before the first paint, so there is no flash of the wrong theme on load. Terminal
|
||||
font family and weight are per device too: a normal and a bold weight, each from 100 to
|
||||
900, and the bundled JetBrains Mono renders every step.
|
||||
|
||||
The same section has the entrance animations for tabs, terminals, agent windows, and
|
||||
lineage lines. All of them default to the legacy no-animation behaviour, so an untouched
|
||||
|
||||
@@ -138,6 +138,12 @@ That is the PTY-exit circuit breaker. Repeated rapid PTY exits trip it, and it b
|
||||
automatic restarts so a broken configuration does not spin forever. Reset it explicitly from
|
||||
the session's controls. Reattaching does not clear it, deliberately.
|
||||
|
||||
### Typed prompts are silently ignored after restoring a tab
|
||||
|
||||
Update. A browser whose input sequence counter fell behind the server's (a restored tab,
|
||||
cleared site data) used to have every prompt deduplicated away. Since 1.29.0 the duplicate
|
||||
acknowledgement carries the watermark and the client re-sends.
|
||||
|
||||
### Sessions I did not create appeared, or my session resized itself
|
||||
|
||||
Two Codeman servers are running against the same data directory and tmux socket. The second
|
||||
@@ -167,6 +173,16 @@ Things to try:
|
||||
Codex ignores the mouse reports that forwarding would send, so Codeman does not forward
|
||||
there. Scrolling is local, and `Shift+Wheel` behaves the same way.
|
||||
|
||||
### Selected text is invisible on a light skin
|
||||
|
||||
Update. Every skin named its selection colour under a key xterm renamed in v5, so the four
|
||||
light skins painted white at 30% over near-white. Fixed in 1.29.0.
|
||||
|
||||
### `Ctrl+Z` suspended my agent
|
||||
|
||||
Update. Since 1.28.0 `Ctrl+Z` is swallowed in agent sessions, so a running CLI cannot be
|
||||
stopped by job control. Shell sessions keep it.
|
||||
|
||||
### `Ctrl+C` copies when I wanted to interrupt
|
||||
|
||||
With a selection, `Ctrl+C` copies. With no selection, it interrupts. Clear the selection
|
||||
@@ -252,11 +268,25 @@ node scripts/build-agent-image.mjs --no-cache
|
||||
A plain rebuild reuses the cached `npm install -g` layer and keeps the CLIs frozen at their
|
||||
original versions while reporting success.
|
||||
|
||||
### Every file in a remote case says "File not found"
|
||||
|
||||
Update. Before 1.29.0 the file routes resolved every path on the Codeman host, so in a
|
||||
remote case every click failed while the file plainly existed on the other machine. Reads
|
||||
now go over ssh; see [Working With Files](Working-With-Files). Editing and Office previews
|
||||
stay unavailable remotely and say so with a 400.
|
||||
|
||||
### Compose: the server crash-loops with `EACCES` on first start
|
||||
|
||||
Start the stack with `bash docker/Start-Codeman.sh` rather than a plain `docker compose up`,
|
||||
and update: since 1.29.0 the entrypoint corrects a root-owned bind mount before dropping
|
||||
privileges. See [Running As A Service](Running-As-A-Service).
|
||||
|
||||
### A remote SSH session dropped and did not come back
|
||||
|
||||
A bounded-backoff watcher reattaches dropped sessions, and it is on by default. Intentional
|
||||
kills are never revived. Check the host is reachable and that the remote tmux server is
|
||||
still running.
|
||||
kills are never revived, and neither is a clean exit inside the pane (Ctrl-D, `exit`): only
|
||||
a transport drop is reconnected. Check the host is reachable and that the remote tmux server
|
||||
is still running.
|
||||
|
||||
## Gathering diagnostics
|
||||
|
||||
|
||||
@@ -24,6 +24,23 @@ Switching tabs does not reload a dashboard. Frames stay alive in the background,
|
||||
took a while to authenticate is still there when you come back. Past six live frames, the
|
||||
least recently viewed is dropped to bound memory.
|
||||
|
||||
## Single-page apps, reloads and links
|
||||
|
||||
A history-routed dashboard (React Router, Vue Router, a Vite dev server) sees the path it
|
||||
would see on its own origin, not the proxy prefix, so it renders its real route instead of
|
||||
its own "page not found". A navigation the page starts itself afterwards, a dev server's
|
||||
full reload or a root-absolute `location.href`, would land outside the proxy with no
|
||||
capability; Codeman recognises it, answers with a small recovery page, and remounts the
|
||||
frame at the path that was lost, bounded to five recoveries a minute per frame. A reload on
|
||||
the dashboard's landing page is recovered the same way.
|
||||
|
||||
A `localhost` or `127.0.0.1` link in agent output opens as a web tab automatically, reusing
|
||||
a saved dashboard for the same server or saving one under its `host:port`. On a phone that
|
||||
address only exists on the Codeman box, so the link would otherwise be a guaranteed
|
||||
connection error. LAN and tailnet addresses still open directly. `*.localhost` names are
|
||||
deliberately not auto-routed: they are DNS names rather than address literals, and the link
|
||||
came from agent output. Add such a dashboard by hand instead.
|
||||
|
||||
## Why dashboards are proxied
|
||||
|
||||
A plain cross-origin iframe fails three ways at once in the setup Codeman actually ships in:
|
||||
@@ -89,6 +106,12 @@ The proxy authenticates on an in-memory capability embedded in the path, which i
|
||||
exempt from the cookie and Origin checks that every API route enforces. That exemption is
|
||||
fenced to safe methods and non-API paths, and there is a test pinning it in place.
|
||||
|
||||
Saved URLs are refused when they point at a link-local or cloud-metadata address, at save
|
||||
time and again against the address the name resolves to at connect time; loopback and
|
||||
private ranges stay allowed, because a `localhost` Grafana is the feature. Capabilities are
|
||||
revoked on logout, and proxied responses carry a same-origin referrer policy so a dashboard
|
||||
cannot hand the capability-bearing URL to a third party.
|
||||
|
||||
Two failure modes that only appear inside a sandboxed frame, and that curl can never
|
||||
reproduce, are handled: runtime-built root-absolute URLs escaping the injected base, and
|
||||
same-host requests being CORS-checked with a null origin. Both present as the dashboard's own
|
||||
|
||||
@@ -110,6 +110,22 @@ it is written. Outside the workspace they open in the preview instead: the tail
|
||||
|
||||
Nothing is registered until you click. Opening a file this way does not add an attachment card.
|
||||
|
||||
## Remote (SSH) cases
|
||||
|
||||
In a remote case the workspace lives on the other machine, and so do the files. Previews,
|
||||
downloads, text reads and the clicked-path route all go over the same ssh connection the
|
||||
session uses: one `realpath` plus `stat` probe for the file and the workspace root, then a
|
||||
streamed `cat` (or a slice of it, so video seeking works). Symlinks are resolved on the host
|
||||
that can resolve them, the size cap applies to the remote size before a byte is requested,
|
||||
and an unreachable host answers 502 rather than pretending the file is missing. Nothing is
|
||||
ever copied onto the Codeman host, and a same-named local file is never served under a
|
||||
remote name.
|
||||
|
||||
Not available over ssh, and said so with a 400 instead of a misleading 404: editing in
|
||||
place, Office previews and generated thumbnails (both need the bytes on the server's disk),
|
||||
the file tree and path picker, and the tail viewer. Docker cases are unaffected, because
|
||||
their workspace is bind-mounted at the same path.
|
||||
|
||||
## The path picker
|
||||
|
||||
For choosing a path rather than typing one. It appears in two places:
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
|
||||
- [The Dashboard](The-Dashboard)
|
||||
- [Agent CLIs](Agent-CLIs)
|
||||
- [Custom Model Endpoints](Custom-Model-Endpoints)
|
||||
- [Working With Files](Working-With-Files)
|
||||
- [Input And Voice](Input-And-Voice)
|
||||
- [Mobile Guide](Mobile-Guide)
|
||||
|
||||
+1506
-590
File diff suppressed because it is too large
Load Diff
Generated
+2
-2
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "aicodeman",
|
||||
"version": "1.29.0",
|
||||
"version": "1.31.0",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "aicodeman",
|
||||
"version": "1.29.0",
|
||||
"version": "1.31.0",
|
||||
"hasInstallScript": true,
|
||||
"license": "MIT",
|
||||
"workspaces": [
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "aicodeman",
|
||||
"version": "1.29.0",
|
||||
"version": "1.31.0",
|
||||
"description": "Mission control for AI coding agents - run 20 autonomous agents with real-time monitoring and session persistence",
|
||||
"type": "module",
|
||||
"main": "dist/index.js",
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "codeman",
|
||||
"description": "Drive Codeman, the self-hosted session manager for AI coding agents, from inside a Claude Code session: spawn worker sessions, prompt them, wait for them, read their answers, clean up. Acts only inside a Codeman-managed session.",
|
||||
"version": "1.29.0",
|
||||
"version": "1.31.0",
|
||||
"author": {
|
||||
"name": "Ark0N",
|
||||
"url": "https://github.com/Ark0N"
|
||||
|
||||
@@ -47,7 +47,7 @@ later call opens with, and your first REAL call performs them anyway:
|
||||
|
||||
```bash
|
||||
. "${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh" 2>/dev/null
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
```
|
||||
|
||||
⚠️ **Never spend a Bash call on this check alone.** §1's block opens with this same
|
||||
@@ -75,8 +75,8 @@ PRE="${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh"
|
||||
mkdir -p "$(dirname "$PRE")"
|
||||
# Rewrite unless the file already ends with THIS version's stamp, so a stale or a
|
||||
# half-written file self-heals here instead of costing you a round trip to rm it.
|
||||
grep -qs '^CODEMAN_PREAMBLE=1.22.0$' "$PRE" || (umask 077; cat > "$PRE" <<'PREAMBLE'
|
||||
# ---- Codeman agent preamble 1.22.0 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
grep -qs '^CODEMAN_PREAMBLE=1.30.1$' "$PRE" || (umask 077; cat > "$PRE" <<'PREAMBLE'
|
||||
# ---- Codeman agent preamble 1.30.1 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
API="${CODEMAN_API_URL:?CODEMAN_API_URL not set; refusing to guess}"
|
||||
SELF="${CODEMAN_SESSION_ID:?CODEMAN_SESSION_ID not set}"
|
||||
# Credentials, cheapest first. Your session has usually INHERITED the server's
|
||||
@@ -150,6 +150,27 @@ _trust_key() { # <sid> -> "confirm" | "move" | "" (nothing safe to press)
|
||||
| tr -d ' \t' | grep -i '❯[0-9.]*\(yes,itrustthisfolder\|no,exit\)' | tail -1 \
|
||||
| sed -e 's/.*[Yy]es,.*/confirm/' -e 's/.*[Nn]o,.*/move/'
|
||||
}
|
||||
# ---- the composer: is the prompt still sitting there, unsent? ----
|
||||
# ⚠️ Claude Code 2.1.277 (auto-installed 2026-09-18) takes typed text the moment the
|
||||
# composer paints but IGNORES Enter for the first 30-50 seconds after it: the \r that
|
||||
# Codeman sends 50 ms after the text and a lone nudge at 20 s both leave the prompt
|
||||
# stranded, with `0 tokens`, while the wait burns its whole timeout. Measured through
|
||||
# this very route: Enter at 28 s stranded, Enter at 51 s submitted. So sendwait READS
|
||||
# the composer and keeps pressing Enter while the prompt is still there.
|
||||
_composer_text() { # <sid> -> the composer's text with ALL whitespace removed: "" once
|
||||
# the prompt was taken, "?" when the pane shows no composer at all. The composer is
|
||||
# the LAST `❯` line: Claude Code echoes a submitted prompt with the same glyph higher
|
||||
# up in the transcript, so only the last one says whether the text was taken.
|
||||
local t
|
||||
t=$("${CURL[@]}" -G "$API/api/v1/sessions/$1/terminal" --data-urlencode 'full=1' \
|
||||
| jq -r '.data.terminalBuffer // empty' \
|
||||
| sed -e "s/$(printf '\033')\[[0-9;?]*[a-zA-Z]//g" -e "s/$(printf '\033')[()][AB0]//g" \
|
||||
| tr -d '\r' | grep -a '^[[:space:]]*❯' | tail -1)
|
||||
[ -n "$t" ] || { printf '?'; return 0; }
|
||||
# Claude Code draws a NO-BREAK SPACE (U+00A0) after the glyph, which [:space:] does
|
||||
# not cover, so it is stripped by its bytes, portably (BSD sed has no \xHH).
|
||||
printf '%s' "$t" | sed 's/^[[:space:]]*❯//' | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g"
|
||||
}
|
||||
_accept_trust() { # <sid> -> 0 once it has answered the dialog, 1 if it could not
|
||||
local sid="$1" k i=1
|
||||
while [ "$i" -le 6 ]; do
|
||||
@@ -262,12 +283,16 @@ spawn_workers() {
|
||||
# worker a silent no-op that still "succeeds" and reports the previous turn's state.
|
||||
# Pass seq explicitly for exactly one reason: resending a possibly-delivered frame as a
|
||||
# deliberate duplicate, at the SAME number (§5.3).
|
||||
# Delivery is SELF-HEALING: an Ink repaint occasionally eats the Enter, leaving the
|
||||
# typed prompt stranded on the composer while a long wait runs its whole timeout
|
||||
# (observed live). So the first wait is short; on its timeout a bare \r goes out (the
|
||||
# missing Enter when the prompt is stranded, a no-op when the turn is genuinely
|
||||
# running), then the ORIGINAL frame is resent unchanged, which the server takes as a
|
||||
# tagged duplicate: it re-waits without retyping (§5.3). Trustworthy for a worker
|
||||
# Delivery is SELF-HEALING: the Enter can be lost (an Ink repaint eats it, and Claude
|
||||
# Code 2.1.277+ ignores it outright for the first 30-50 s after the composer paints),
|
||||
# leaving the typed prompt stranded on the composer while a long wait runs its whole
|
||||
# timeout (observed live, twelve reviews in a row). So the first wait is short; on its
|
||||
# timeout the ORIGINAL frame is resent unchanged as a long re-wait (a tagged duplicate:
|
||||
# the server re-waits without retyping, §5.3) and kept open in the background, while
|
||||
# the composer is READ (_composer_text) and, as long as the prompt is still sitting
|
||||
# there, a bare \r goes out about every ten seconds, up to twelve times. An empty
|
||||
# composer ends the loop, so a prompt that was taken is never nudged again, and the
|
||||
# wait that was open the whole time is what reports the turn's end. Trustworthy for a worker
|
||||
# spawn_worker handed back -- claude (hooks vetted) or deepseek (status bridge) --
|
||||
# and for those only. Hook-less workspaces and the other modes resolve on flapping
|
||||
# idle: markers instead (§5.5). ⚠️ A dsh worker running a profile that does not
|
||||
@@ -275,7 +300,7 @@ spawn_workers() {
|
||||
# it accepts the send and then burns both waits. One timeout on a dsh worker whose
|
||||
# pane clearly finished means that profile, so switch that worker to markers.
|
||||
sendwait() {
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r c head n=0 tmp bg i
|
||||
# `wait:"stop,exit"`, never the `wait:true` default set: that set also carries
|
||||
# `idle`, which is INFERRED from output stabilization and flaps mid-turn. On a
|
||||
# dsh worker whose TUI repaints rarely the session reads `idle` while the model
|
||||
@@ -290,16 +315,38 @@ sendwait() {
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$body")
|
||||
if jq -e '.data.delivered and .data.wait.timedOut' <<<"$r" >/dev/null 2>&1; then
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
# The resend is a tagged DUPLICATE, so the server skips the write and reports
|
||||
# `delivered:false` for it -- truthfully, but about the wrong send. The first
|
||||
# one delivered, so carry that forward, or §1's cleanup reads a completed turn
|
||||
# as an undelivered one and keeps a finished worker forever.
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" \
|
||||
| jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end')
|
||||
# ⚠️ The long re-wait is registered FIRST and stays open for the rest of this call,
|
||||
# in the background, while the Enter loop below works the composer. Signals have
|
||||
# no history: a `stop` that fires while no wait is open (during a composer read
|
||||
# between two short waits, measured) is lost, and the next wait then runs its
|
||||
# whole timeout on a turn that already ended. The resend is a tagged DUPLICATE,
|
||||
# so the server skips the write and re-waits without retyping (§5.3).
|
||||
tmp=$(mktemp "${TMPDIR:-/tmp}/codeman-wait.XXXXXX") || return 1
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" > "$tmp" &
|
||||
bg=$!
|
||||
# The prompt's head with whitespace removed, matched literally (the "$head"
|
||||
# quoting inside ${c#...} keeps a * or ? in the prompt from acting as a glob).
|
||||
head=$(printf '%s' "$p" | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g" | head -c 24)
|
||||
while [ "$n" -lt 12 ] && [ ! -s "$tmp" ]; do # a non-empty file means the wait ended
|
||||
c=$(_composer_text "$sid")
|
||||
if [ "$c" = '?' ]; then
|
||||
[ "$n" -eq 0 ] || break # unreadable pane: one Enter, then trust it
|
||||
elif [ -z "$head" ] || [ "${c#"$head"}" = "$c" ]; then
|
||||
break # composer empty (taken) or holding other text
|
||||
fi
|
||||
n=$((n+1))
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
i=0; while [ "$i" -lt 10 ] && [ ! -s "$tmp" ]; do sleep 1; i=$((i+1)); done
|
||||
done
|
||||
wait "$bg"
|
||||
# The duplicate reports `delivered:false` -- truthfully, but about the wrong send.
|
||||
# The first one delivered, so carry that forward, or §1's cleanup reads a completed
|
||||
# turn as an undelivered one and keeps a finished worker forever.
|
||||
r=$(jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end' < "$tmp")
|
||||
rm -f "$tmp"
|
||||
fi
|
||||
printf '%s\n' "$r"
|
||||
}
|
||||
@@ -325,10 +372,10 @@ last_text() {
|
||||
# The stamp is the LAST line on purpose (a truncated write leaves it unset) and is kept
|
||||
# bare on purpose: the write condition above anchors on it with $, so an inline comment
|
||||
# here would fail that match and rewrite this file on every single bootstrap.
|
||||
CODEMAN_PREAMBLE=1.22.0
|
||||
CODEMAN_PREAMBLE=1.30.1
|
||||
PREAMBLE
|
||||
)
|
||||
. "$PRE"; [ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble at $PRE is stale or truncated: rm it and re-run this block"; exit 1; }
|
||||
. "$PRE"; [ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble at $PRE is stale or truncated: rm it and re-run this block"; exit 1; }
|
||||
```
|
||||
|
||||
Every later Bash call that touches the API starts with the same two loader lines from
|
||||
@@ -379,7 +426,7 @@ and no per-call body to hand-build.
|
||||
|
||||
```bash
|
||||
. "${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh" 2>/dev/null # §0 loader
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
N=(alpha beta) # INVENT one fresh case name per worker; never list cases first
|
||||
# (a name may carry a mode: `beta:deepseek`, see below)
|
||||
T=('reply with one line: the absolute path of your working directory'
|
||||
@@ -440,9 +487,11 @@ Four things this block leans on, each one link away, no detour needed to run it:
|
||||
skill: §5.1. Those workspaces do get hooks now, unless the operator disabled it.
|
||||
- `sendwait` supplies the `\r`, picks a fresh `seq`, and self-heals a stranded Enter.
|
||||
A prompt without the `\r` is never submitted (§3), a reused `seq` is silently
|
||||
swallowed as an already-applied duplicate, and an Enter eaten by an Ink repaint
|
||||
strands the prompt on the composer until a bare `\r` follows: all three are reasons
|
||||
to let `sendwait` build the call rather than hand-rolling it.
|
||||
swallowed as an already-applied duplicate, and a lost Enter strands the prompt on the
|
||||
composer until a bare `\r` follows: Claude Code 2.1.277 and later ignore Enter for the
|
||||
first 30 to 50 seconds after the composer paints while still taking the text, so
|
||||
`sendwait` reads the composer and keeps pressing Enter until the prompt has left it.
|
||||
All three are reasons to let `sendwait` build the call rather than hand-rolling it.
|
||||
- Each `sendwait` costs that worker one billed turn, as does every prompt you send it.
|
||||
- Deleting the sessions does **not** remove the case directories. They are marked as
|
||||
agent-created, so `GET /api/v1/cases/agent-created` lists them for cleanup: §5.14.
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# ---- Codeman agent preamble 1.22.0 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
# ---- Codeman agent preamble 1.30.1 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
API="${CODEMAN_API_URL:?CODEMAN_API_URL not set; refusing to guess}"
|
||||
SELF="${CODEMAN_SESSION_ID:?CODEMAN_SESSION_ID not set}"
|
||||
# Credentials, cheapest first. Your session has usually INHERITED the server's
|
||||
@@ -72,6 +72,27 @@ _trust_key() { # <sid> -> "confirm" | "move" | "" (nothing safe to press)
|
||||
| tr -d ' \t' | grep -i '❯[0-9.]*\(yes,itrustthisfolder\|no,exit\)' | tail -1 \
|
||||
| sed -e 's/.*[Yy]es,.*/confirm/' -e 's/.*[Nn]o,.*/move/'
|
||||
}
|
||||
# ---- the composer: is the prompt still sitting there, unsent? ----
|
||||
# ⚠️ Claude Code 2.1.277 (auto-installed 2026-09-18) takes typed text the moment the
|
||||
# composer paints but IGNORES Enter for the first 30-50 seconds after it: the \r that
|
||||
# Codeman sends 50 ms after the text and a lone nudge at 20 s both leave the prompt
|
||||
# stranded, with `0 tokens`, while the wait burns its whole timeout. Measured through
|
||||
# this very route: Enter at 28 s stranded, Enter at 51 s submitted. So sendwait READS
|
||||
# the composer and keeps pressing Enter while the prompt is still there.
|
||||
_composer_text() { # <sid> -> the composer's text with ALL whitespace removed: "" once
|
||||
# the prompt was taken, "?" when the pane shows no composer at all. The composer is
|
||||
# the LAST `❯` line: Claude Code echoes a submitted prompt with the same glyph higher
|
||||
# up in the transcript, so only the last one says whether the text was taken.
|
||||
local t
|
||||
t=$("${CURL[@]}" -G "$API/api/v1/sessions/$1/terminal" --data-urlencode 'full=1' \
|
||||
| jq -r '.data.terminalBuffer // empty' \
|
||||
| sed -e "s/$(printf '\033')\[[0-9;?]*[a-zA-Z]//g" -e "s/$(printf '\033')[()][AB0]//g" \
|
||||
| tr -d '\r' | grep -a '^[[:space:]]*❯' | tail -1)
|
||||
[ -n "$t" ] || { printf '?'; return 0; }
|
||||
# Claude Code draws a NO-BREAK SPACE (U+00A0) after the glyph, which [:space:] does
|
||||
# not cover, so it is stripped by its bytes, portably (BSD sed has no \xHH).
|
||||
printf '%s' "$t" | sed 's/^[[:space:]]*❯//' | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g"
|
||||
}
|
||||
_accept_trust() { # <sid> -> 0 once it has answered the dialog, 1 if it could not
|
||||
local sid="$1" k i=1
|
||||
while [ "$i" -le 6 ]; do
|
||||
@@ -184,12 +205,16 @@ spawn_workers() {
|
||||
# worker a silent no-op that still "succeeds" and reports the previous turn's state.
|
||||
# Pass seq explicitly for exactly one reason: resending a possibly-delivered frame as a
|
||||
# deliberate duplicate, at the SAME number (§5.3).
|
||||
# Delivery is SELF-HEALING: an Ink repaint occasionally eats the Enter, leaving the
|
||||
# typed prompt stranded on the composer while a long wait runs its whole timeout
|
||||
# (observed live). So the first wait is short; on its timeout a bare \r goes out (the
|
||||
# missing Enter when the prompt is stranded, a no-op when the turn is genuinely
|
||||
# running), then the ORIGINAL frame is resent unchanged, which the server takes as a
|
||||
# tagged duplicate: it re-waits without retyping (§5.3). Trustworthy for a worker
|
||||
# Delivery is SELF-HEALING: the Enter can be lost (an Ink repaint eats it, and Claude
|
||||
# Code 2.1.277+ ignores it outright for the first 30-50 s after the composer paints),
|
||||
# leaving the typed prompt stranded on the composer while a long wait runs its whole
|
||||
# timeout (observed live, twelve reviews in a row). So the first wait is short; on its
|
||||
# timeout the ORIGINAL frame is resent unchanged as a long re-wait (a tagged duplicate:
|
||||
# the server re-waits without retyping, §5.3) and kept open in the background, while
|
||||
# the composer is READ (_composer_text) and, as long as the prompt is still sitting
|
||||
# there, a bare \r goes out about every ten seconds, up to twelve times. An empty
|
||||
# composer ends the loop, so a prompt that was taken is never nudged again, and the
|
||||
# wait that was open the whole time is what reports the turn's end. Trustworthy for a worker
|
||||
# spawn_worker handed back -- claude (hooks vetted) or deepseek (status bridge) --
|
||||
# and for those only. Hook-less workspaces and the other modes resolve on flapping
|
||||
# idle: markers instead (§5.5). ⚠️ A dsh worker running a profile that does not
|
||||
@@ -197,7 +222,7 @@ spawn_workers() {
|
||||
# it accepts the send and then burns both waits. One timeout on a dsh worker whose
|
||||
# pane clearly finished means that profile, so switch that worker to markers.
|
||||
sendwait() {
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r c head n=0 tmp bg i
|
||||
# `wait:"stop,exit"`, never the `wait:true` default set: that set also carries
|
||||
# `idle`, which is INFERRED from output stabilization and flaps mid-turn. On a
|
||||
# dsh worker whose TUI repaints rarely the session reads `idle` while the model
|
||||
@@ -212,16 +237,38 @@ sendwait() {
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$body")
|
||||
if jq -e '.data.delivered and .data.wait.timedOut' <<<"$r" >/dev/null 2>&1; then
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
# The resend is a tagged DUPLICATE, so the server skips the write and reports
|
||||
# `delivered:false` for it -- truthfully, but about the wrong send. The first
|
||||
# one delivered, so carry that forward, or §1's cleanup reads a completed turn
|
||||
# as an undelivered one and keeps a finished worker forever.
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" \
|
||||
| jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end')
|
||||
# ⚠️ The long re-wait is registered FIRST and stays open for the rest of this call,
|
||||
# in the background, while the Enter loop below works the composer. Signals have
|
||||
# no history: a `stop` that fires while no wait is open (during a composer read
|
||||
# between two short waits, measured) is lost, and the next wait then runs its
|
||||
# whole timeout on a turn that already ended. The resend is a tagged DUPLICATE,
|
||||
# so the server skips the write and re-waits without retyping (§5.3).
|
||||
tmp=$(mktemp "${TMPDIR:-/tmp}/codeman-wait.XXXXXX") || return 1
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" > "$tmp" &
|
||||
bg=$!
|
||||
# The prompt's head with whitespace removed, matched literally (the "$head"
|
||||
# quoting inside ${c#...} keeps a * or ? in the prompt from acting as a glob).
|
||||
head=$(printf '%s' "$p" | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g" | head -c 24)
|
||||
while [ "$n" -lt 12 ] && [ ! -s "$tmp" ]; do # a non-empty file means the wait ended
|
||||
c=$(_composer_text "$sid")
|
||||
if [ "$c" = '?' ]; then
|
||||
[ "$n" -eq 0 ] || break # unreadable pane: one Enter, then trust it
|
||||
elif [ -z "$head" ] || [ "${c#"$head"}" = "$c" ]; then
|
||||
break # composer empty (taken) or holding other text
|
||||
fi
|
||||
n=$((n+1))
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
i=0; while [ "$i" -lt 10 ] && [ ! -s "$tmp" ]; do sleep 1; i=$((i+1)); done
|
||||
done
|
||||
wait "$bg"
|
||||
# The duplicate reports `delivered:false` -- truthfully, but about the wrong send.
|
||||
# The first one delivered, so carry that forward, or §1's cleanup reads a completed
|
||||
# turn as an undelivered one and keeps a finished worker forever.
|
||||
r=$(jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end' < "$tmp")
|
||||
rm -f "$tmp"
|
||||
fi
|
||||
printf '%s\n' "$r"
|
||||
}
|
||||
@@ -247,4 +294,4 @@ last_text() {
|
||||
# The stamp is the LAST line on purpose (a truncated write leaves it unset) and is kept
|
||||
# bare on purpose: the write condition above anchors on it with $, so an inline comment
|
||||
# here would fail that match and rewrite this file on every single bootstrap.
|
||||
CODEMAN_PREAMBLE=1.22.0
|
||||
CODEMAN_PREAMBLE=1.30.1
|
||||
|
||||
@@ -21,7 +21,7 @@ by sourcing the preamble file the §0 bootstrap wrote, and checking its version
|
||||
|
||||
```bash
|
||||
. "${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh" 2>/dev/null
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble missing or stale; re-run the §0 bootstrap"; exit 1; }
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble missing or stale; re-run the §0 bootstrap"; exit 1; }
|
||||
```
|
||||
|
||||
Do **not** re-paste the preamble body into each call. Sourcing it is what retires the
|
||||
|
||||
@@ -135,8 +135,7 @@ export function renderInstallShBlock(entries: CliEntry[] = STOCK_CLIS): string {
|
||||
const ids: string[] = [];
|
||||
const labels: string[] = [];
|
||||
const enabled: string[] = [];
|
||||
const kinds: string[] = [];
|
||||
const npm: string[] = [];
|
||||
const launcherOnly: string[] = [];
|
||||
const docs: string[] = [];
|
||||
const cmdLinux: string[] = [];
|
||||
const cmdDarwin: string[] = [];
|
||||
@@ -151,8 +150,11 @@ export function renderInstallShBlock(entries: CliEntry[] = STOCK_CLIS): string {
|
||||
ids.push(shQuote(entry.id as string));
|
||||
labels.push(shQuote(entry.label));
|
||||
enabled.push(entry.enabled ? '1' : '0');
|
||||
kinds.push(shQuote(entry.kind));
|
||||
npm.push(shQuote(entry.discovery.install.npmPackage ?? ''));
|
||||
// Parallel to CLI_IDS: 1 when this entry's install command installs a launcher rather
|
||||
// than something that can drive a pane on its own (see installCommandFor above). Purely
|
||||
// derived from discovery.launcherProfile — install.sh's hint printer reads this to add a
|
||||
// caveat instead of hardcoding which id it means.
|
||||
launcherOnly.push(entry.discovery.launcherProfile ? '1' : '0');
|
||||
docs.push(shQuote(entry.discovery.install.docsUrl ?? ''));
|
||||
cmdLinux.push(shQuote(installCommandFor(entry, 'linux')));
|
||||
cmdDarwin.push(shQuote(installCommandFor(entry, 'darwin')));
|
||||
@@ -195,8 +197,7 @@ export function renderInstallShBlock(entries: CliEntry[] = STOCK_CLIS): string {
|
||||
arr('CLI_IDS', ids),
|
||||
arr('CLI_LABELS', labels),
|
||||
arr('CLI_ENABLED', enabled),
|
||||
arr('CLI_KIND', kinds),
|
||||
arr('CLI_NPM', npm),
|
||||
arr('CLI_LAUNCHER_ONLY', launcherOnly),
|
||||
arr('CLI_DOCS', docs),
|
||||
arr('CLI_CMD_LINUX', cmdLinux),
|
||||
arr('CLI_CMD_DARWIN', cmdDarwin),
|
||||
|
||||
+75
-26
@@ -47,7 +47,7 @@ later call opens with, and your first REAL call performs them anyway:
|
||||
|
||||
```bash
|
||||
. "${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh" 2>/dev/null
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
```
|
||||
|
||||
⚠️ **Never spend a Bash call on this check alone.** §1's block opens with this same
|
||||
@@ -75,8 +75,8 @@ PRE="${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh"
|
||||
mkdir -p "$(dirname "$PRE")"
|
||||
# Rewrite unless the file already ends with THIS version's stamp, so a stale or a
|
||||
# half-written file self-heals here instead of costing you a round trip to rm it.
|
||||
grep -qs '^CODEMAN_PREAMBLE=1.22.0$' "$PRE" || (umask 077; cat > "$PRE" <<'PREAMBLE'
|
||||
# ---- Codeman agent preamble 1.22.0 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
grep -qs '^CODEMAN_PREAMBLE=1.30.1$' "$PRE" || (umask 077; cat > "$PRE" <<'PREAMBLE'
|
||||
# ---- Codeman agent preamble 1.30.1 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
API="${CODEMAN_API_URL:?CODEMAN_API_URL not set; refusing to guess}"
|
||||
SELF="${CODEMAN_SESSION_ID:?CODEMAN_SESSION_ID not set}"
|
||||
# Credentials, cheapest first. Your session has usually INHERITED the server's
|
||||
@@ -150,6 +150,27 @@ _trust_key() { # <sid> -> "confirm" | "move" | "" (nothing safe to press)
|
||||
| tr -d ' \t' | grep -i '❯[0-9.]*\(yes,itrustthisfolder\|no,exit\)' | tail -1 \
|
||||
| sed -e 's/.*[Yy]es,.*/confirm/' -e 's/.*[Nn]o,.*/move/'
|
||||
}
|
||||
# ---- the composer: is the prompt still sitting there, unsent? ----
|
||||
# ⚠️ Claude Code 2.1.277 (auto-installed 2026-09-18) takes typed text the moment the
|
||||
# composer paints but IGNORES Enter for the first 30-50 seconds after it: the \r that
|
||||
# Codeman sends 50 ms after the text and a lone nudge at 20 s both leave the prompt
|
||||
# stranded, with `0 tokens`, while the wait burns its whole timeout. Measured through
|
||||
# this very route: Enter at 28 s stranded, Enter at 51 s submitted. So sendwait READS
|
||||
# the composer and keeps pressing Enter while the prompt is still there.
|
||||
_composer_text() { # <sid> -> the composer's text with ALL whitespace removed: "" once
|
||||
# the prompt was taken, "?" when the pane shows no composer at all. The composer is
|
||||
# the LAST `❯` line: Claude Code echoes a submitted prompt with the same glyph higher
|
||||
# up in the transcript, so only the last one says whether the text was taken.
|
||||
local t
|
||||
t=$("${CURL[@]}" -G "$API/api/v1/sessions/$1/terminal" --data-urlencode 'full=1' \
|
||||
| jq -r '.data.terminalBuffer // empty' \
|
||||
| sed -e "s/$(printf '\033')\[[0-9;?]*[a-zA-Z]//g" -e "s/$(printf '\033')[()][AB0]//g" \
|
||||
| tr -d '\r' | grep -a '^[[:space:]]*❯' | tail -1)
|
||||
[ -n "$t" ] || { printf '?'; return 0; }
|
||||
# Claude Code draws a NO-BREAK SPACE (U+00A0) after the glyph, which [:space:] does
|
||||
# not cover, so it is stripped by its bytes, portably (BSD sed has no \xHH).
|
||||
printf '%s' "$t" | sed 's/^[[:space:]]*❯//' | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g"
|
||||
}
|
||||
_accept_trust() { # <sid> -> 0 once it has answered the dialog, 1 if it could not
|
||||
local sid="$1" k i=1
|
||||
while [ "$i" -le 6 ]; do
|
||||
@@ -262,12 +283,16 @@ spawn_workers() {
|
||||
# worker a silent no-op that still "succeeds" and reports the previous turn's state.
|
||||
# Pass seq explicitly for exactly one reason: resending a possibly-delivered frame as a
|
||||
# deliberate duplicate, at the SAME number (§5.3).
|
||||
# Delivery is SELF-HEALING: an Ink repaint occasionally eats the Enter, leaving the
|
||||
# typed prompt stranded on the composer while a long wait runs its whole timeout
|
||||
# (observed live). So the first wait is short; on its timeout a bare \r goes out (the
|
||||
# missing Enter when the prompt is stranded, a no-op when the turn is genuinely
|
||||
# running), then the ORIGINAL frame is resent unchanged, which the server takes as a
|
||||
# tagged duplicate: it re-waits without retyping (§5.3). Trustworthy for a worker
|
||||
# Delivery is SELF-HEALING: the Enter can be lost (an Ink repaint eats it, and Claude
|
||||
# Code 2.1.277+ ignores it outright for the first 30-50 s after the composer paints),
|
||||
# leaving the typed prompt stranded on the composer while a long wait runs its whole
|
||||
# timeout (observed live, twelve reviews in a row). So the first wait is short; on its
|
||||
# timeout the ORIGINAL frame is resent unchanged as a long re-wait (a tagged duplicate:
|
||||
# the server re-waits without retyping, §5.3) and kept open in the background, while
|
||||
# the composer is READ (_composer_text) and, as long as the prompt is still sitting
|
||||
# there, a bare \r goes out about every ten seconds, up to twelve times. An empty
|
||||
# composer ends the loop, so a prompt that was taken is never nudged again, and the
|
||||
# wait that was open the whole time is what reports the turn's end. Trustworthy for a worker
|
||||
# spawn_worker handed back -- claude (hooks vetted) or deepseek (status bridge) --
|
||||
# and for those only. Hook-less workspaces and the other modes resolve on flapping
|
||||
# idle: markers instead (§5.5). ⚠️ A dsh worker running a profile that does not
|
||||
@@ -275,7 +300,7 @@ spawn_workers() {
|
||||
# it accepts the send and then burns both waits. One timeout on a dsh worker whose
|
||||
# pane clearly finished means that profile, so switch that worker to markers.
|
||||
sendwait() {
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r c head n=0 tmp bg i
|
||||
# `wait:"stop,exit"`, never the `wait:true` default set: that set also carries
|
||||
# `idle`, which is INFERRED from output stabilization and flaps mid-turn. On a
|
||||
# dsh worker whose TUI repaints rarely the session reads `idle` while the model
|
||||
@@ -290,16 +315,38 @@ sendwait() {
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$body")
|
||||
if jq -e '.data.delivered and .data.wait.timedOut' <<<"$r" >/dev/null 2>&1; then
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
# The resend is a tagged DUPLICATE, so the server skips the write and reports
|
||||
# `delivered:false` for it -- truthfully, but about the wrong send. The first
|
||||
# one delivered, so carry that forward, or §1's cleanup reads a completed turn
|
||||
# as an undelivered one and keeps a finished worker forever.
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" \
|
||||
| jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end')
|
||||
# ⚠️ The long re-wait is registered FIRST and stays open for the rest of this call,
|
||||
# in the background, while the Enter loop below works the composer. Signals have
|
||||
# no history: a `stop` that fires while no wait is open (during a composer read
|
||||
# between two short waits, measured) is lost, and the next wait then runs its
|
||||
# whole timeout on a turn that already ended. The resend is a tagged DUPLICATE,
|
||||
# so the server skips the write and re-waits without retyping (§5.3).
|
||||
tmp=$(mktemp "${TMPDIR:-/tmp}/codeman-wait.XXXXXX") || return 1
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" > "$tmp" &
|
||||
bg=$!
|
||||
# The prompt's head with whitespace removed, matched literally (the "$head"
|
||||
# quoting inside ${c#...} keeps a * or ? in the prompt from acting as a glob).
|
||||
head=$(printf '%s' "$p" | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g" | head -c 24)
|
||||
while [ "$n" -lt 12 ] && [ ! -s "$tmp" ]; do # a non-empty file means the wait ended
|
||||
c=$(_composer_text "$sid")
|
||||
if [ "$c" = '?' ]; then
|
||||
[ "$n" -eq 0 ] || break # unreadable pane: one Enter, then trust it
|
||||
elif [ -z "$head" ] || [ "${c#"$head"}" = "$c" ]; then
|
||||
break # composer empty (taken) or holding other text
|
||||
fi
|
||||
n=$((n+1))
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
i=0; while [ "$i" -lt 10 ] && [ ! -s "$tmp" ]; do sleep 1; i=$((i+1)); done
|
||||
done
|
||||
wait "$bg"
|
||||
# The duplicate reports `delivered:false` -- truthfully, but about the wrong send.
|
||||
# The first one delivered, so carry that forward, or §1's cleanup reads a completed
|
||||
# turn as an undelivered one and keeps a finished worker forever.
|
||||
r=$(jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end' < "$tmp")
|
||||
rm -f "$tmp"
|
||||
fi
|
||||
printf '%s\n' "$r"
|
||||
}
|
||||
@@ -325,10 +372,10 @@ last_text() {
|
||||
# The stamp is the LAST line on purpose (a truncated write leaves it unset) and is kept
|
||||
# bare on purpose: the write condition above anchors on it with $, so an inline comment
|
||||
# here would fail that match and rewrite this file on every single bootstrap.
|
||||
CODEMAN_PREAMBLE=1.22.0
|
||||
CODEMAN_PREAMBLE=1.30.1
|
||||
PREAMBLE
|
||||
)
|
||||
. "$PRE"; [ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble at $PRE is stale or truncated: rm it and re-run this block"; exit 1; }
|
||||
. "$PRE"; [ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble at $PRE is stale or truncated: rm it and re-run this block"; exit 1; }
|
||||
```
|
||||
|
||||
Every later Bash call that touches the API starts with the same two loader lines from
|
||||
@@ -379,7 +426,7 @@ and no per-call body to hand-build.
|
||||
|
||||
```bash
|
||||
. "${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh" 2>/dev/null # §0 loader
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble missing or stale; run the full §0 block"; exit 1; }
|
||||
N=(alpha beta) # INVENT one fresh case name per worker; never list cases first
|
||||
# (a name may carry a mode: `beta:deepseek`, see below)
|
||||
T=('reply with one line: the absolute path of your working directory'
|
||||
@@ -440,9 +487,11 @@ Four things this block leans on, each one link away, no detour needed to run it:
|
||||
skill: §5.1. Those workspaces do get hooks now, unless the operator disabled it.
|
||||
- `sendwait` supplies the `\r`, picks a fresh `seq`, and self-heals a stranded Enter.
|
||||
A prompt without the `\r` is never submitted (§3), a reused `seq` is silently
|
||||
swallowed as an already-applied duplicate, and an Enter eaten by an Ink repaint
|
||||
strands the prompt on the composer until a bare `\r` follows: all three are reasons
|
||||
to let `sendwait` build the call rather than hand-rolling it.
|
||||
swallowed as an already-applied duplicate, and a lost Enter strands the prompt on the
|
||||
composer until a bare `\r` follows: Claude Code 2.1.277 and later ignore Enter for the
|
||||
first 30 to 50 seconds after the composer paints while still taking the text, so
|
||||
`sendwait` reads the composer and keeps pressing Enter until the prompt has left it.
|
||||
All three are reasons to let `sendwait` build the call rather than hand-rolling it.
|
||||
- Each `sendwait` costs that worker one billed turn, as does every prompt you send it.
|
||||
- Deleting the sessions does **not** remove the case directories. They are marked as
|
||||
agent-created, so `GET /api/v1/cases/agent-created` lists them for cleanup: §5.14.
|
||||
|
||||
+66
-19
@@ -1,4 +1,4 @@
|
||||
# ---- Codeman agent preamble 1.22.0 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
# ---- Codeman agent preamble 1.30.1 (seeded by Codeman at session spawn; the SKILL.md §0 bootstrap rewrites it when missing or stale) ----
|
||||
API="${CODEMAN_API_URL:?CODEMAN_API_URL not set; refusing to guess}"
|
||||
SELF="${CODEMAN_SESSION_ID:?CODEMAN_SESSION_ID not set}"
|
||||
# Credentials, cheapest first. Your session has usually INHERITED the server's
|
||||
@@ -72,6 +72,27 @@ _trust_key() { # <sid> -> "confirm" | "move" | "" (nothing safe to press)
|
||||
| tr -d ' \t' | grep -i '❯[0-9.]*\(yes,itrustthisfolder\|no,exit\)' | tail -1 \
|
||||
| sed -e 's/.*[Yy]es,.*/confirm/' -e 's/.*[Nn]o,.*/move/'
|
||||
}
|
||||
# ---- the composer: is the prompt still sitting there, unsent? ----
|
||||
# ⚠️ Claude Code 2.1.277 (auto-installed 2026-09-18) takes typed text the moment the
|
||||
# composer paints but IGNORES Enter for the first 30-50 seconds after it: the \r that
|
||||
# Codeman sends 50 ms after the text and a lone nudge at 20 s both leave the prompt
|
||||
# stranded, with `0 tokens`, while the wait burns its whole timeout. Measured through
|
||||
# this very route: Enter at 28 s stranded, Enter at 51 s submitted. So sendwait READS
|
||||
# the composer and keeps pressing Enter while the prompt is still there.
|
||||
_composer_text() { # <sid> -> the composer's text with ALL whitespace removed: "" once
|
||||
# the prompt was taken, "?" when the pane shows no composer at all. The composer is
|
||||
# the LAST `❯` line: Claude Code echoes a submitted prompt with the same glyph higher
|
||||
# up in the transcript, so only the last one says whether the text was taken.
|
||||
local t
|
||||
t=$("${CURL[@]}" -G "$API/api/v1/sessions/$1/terminal" --data-urlencode 'full=1' \
|
||||
| jq -r '.data.terminalBuffer // empty' \
|
||||
| sed -e "s/$(printf '\033')\[[0-9;?]*[a-zA-Z]//g" -e "s/$(printf '\033')[()][AB0]//g" \
|
||||
| tr -d '\r' | grep -a '^[[:space:]]*❯' | tail -1)
|
||||
[ -n "$t" ] || { printf '?'; return 0; }
|
||||
# Claude Code draws a NO-BREAK SPACE (U+00A0) after the glyph, which [:space:] does
|
||||
# not cover, so it is stripped by its bytes, portably (BSD sed has no \xHH).
|
||||
printf '%s' "$t" | sed 's/^[[:space:]]*❯//' | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g"
|
||||
}
|
||||
_accept_trust() { # <sid> -> 0 once it has answered the dialog, 1 if it could not
|
||||
local sid="$1" k i=1
|
||||
while [ "$i" -le 6 ]; do
|
||||
@@ -184,12 +205,16 @@ spawn_workers() {
|
||||
# worker a silent no-op that still "succeeds" and reports the previous turn's state.
|
||||
# Pass seq explicitly for exactly one reason: resending a possibly-delivered frame as a
|
||||
# deliberate duplicate, at the SAME number (§5.3).
|
||||
# Delivery is SELF-HEALING: an Ink repaint occasionally eats the Enter, leaving the
|
||||
# typed prompt stranded on the composer while a long wait runs its whole timeout
|
||||
# (observed live). So the first wait is short; on its timeout a bare \r goes out (the
|
||||
# missing Enter when the prompt is stranded, a no-op when the turn is genuinely
|
||||
# running), then the ORIGINAL frame is resent unchanged, which the server takes as a
|
||||
# tagged duplicate: it re-waits without retyping (§5.3). Trustworthy for a worker
|
||||
# Delivery is SELF-HEALING: the Enter can be lost (an Ink repaint eats it, and Claude
|
||||
# Code 2.1.277+ ignores it outright for the first 30-50 s after the composer paints),
|
||||
# leaving the typed prompt stranded on the composer while a long wait runs its whole
|
||||
# timeout (observed live, twelve reviews in a row). So the first wait is short; on its
|
||||
# timeout the ORIGINAL frame is resent unchanged as a long re-wait (a tagged duplicate:
|
||||
# the server re-waits without retyping, §5.3) and kept open in the background, while
|
||||
# the composer is READ (_composer_text) and, as long as the prompt is still sitting
|
||||
# there, a bare \r goes out about every ten seconds, up to twelve times. An empty
|
||||
# composer ends the loop, so a prompt that was taken is never nudged again, and the
|
||||
# wait that was open the whole time is what reports the turn's end. Trustworthy for a worker
|
||||
# spawn_worker handed back -- claude (hooks vetted) or deepseek (status bridge) --
|
||||
# and for those only. Hook-less workspaces and the other modes resolve on flapping
|
||||
# idle: markers instead (§5.5). ⚠️ A dsh worker running a profile that does not
|
||||
@@ -197,7 +222,7 @@ spawn_workers() {
|
||||
# it accepts the send and then burns both waits. One timeout on a dsh worker whose
|
||||
# pane clearly finished means that profile, so switch that worker to markers.
|
||||
sendwait() {
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r
|
||||
local sid="${1:?}" p="${2:?}" seq="${3:-$(date +%s)}" body r c head n=0 tmp bg i
|
||||
# `wait:"stop,exit"`, never the `wait:true` default set: that set also carries
|
||||
# `idle`, which is INFERRED from output stabilization and flaps mid-turn. On a
|
||||
# dsh worker whose TUI repaints rarely the session reads `idle` while the model
|
||||
@@ -212,16 +237,38 @@ sendwait() {
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$body")
|
||||
if jq -e '.data.delivered and .data.wait.timedOut' <<<"$r" >/dev/null 2>&1; then
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
# The resend is a tagged DUPLICATE, so the server skips the write and reports
|
||||
# `delivered:false` for it -- truthfully, but about the wrong send. The first
|
||||
# one delivered, so carry that forward, or §1's cleanup reads a completed turn
|
||||
# as an undelivered one and keeps a finished worker forever.
|
||||
r=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" \
|
||||
| jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end')
|
||||
# ⚠️ The long re-wait is registered FIRST and stays open for the rest of this call,
|
||||
# in the background, while the Enter loop below works the composer. Signals have
|
||||
# no history: a `stop` that fires while no wait is open (during a composer read
|
||||
# between two short waits, measured) is lost, and the next wait then runs its
|
||||
# whole timeout on a turn that already ended. The resend is a tagged DUPLICATE,
|
||||
# so the server skips the write and re-waits without retyping (§5.3).
|
||||
tmp=$(mktemp "${TMPDIR:-/tmp}/codeman-wait.XXXXXX") || return 1
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" \
|
||||
-H 'Content-Type: application/json' --data-binary "$(jq -c '.waitTimeout=580000' <<<"$body")" > "$tmp" &
|
||||
bg=$!
|
||||
# The prompt's head with whitespace removed, matched literally (the "$head"
|
||||
# quoting inside ${c#...} keeps a * or ? in the prompt from acting as a glob).
|
||||
head=$(printf '%s' "$p" | tr -d '[:space:]' | sed "s/$(printf '\302\240')//g" | head -c 24)
|
||||
while [ "$n" -lt 12 ] && [ ! -s "$tmp" ]; do # a non-empty file means the wait ended
|
||||
c=$(_composer_text "$sid")
|
||||
if [ "$c" = '?' ]; then
|
||||
[ "$n" -eq 0 ] || break # unreadable pane: one Enter, then trust it
|
||||
elif [ -z "$head" ] || [ "${c#"$head"}" = "$c" ]; then
|
||||
break # composer empty (taken) or holding other text
|
||||
fi
|
||||
n=$((n+1))
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$sid/input" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -nc --arg c "$CID-$sid" --argjson s "$(date +%s)" \
|
||||
'{input:"\r",useMux:true,clientId:$c,seq:$s}')" >/dev/null
|
||||
i=0; while [ "$i" -lt 10 ] && [ ! -s "$tmp" ]; do sleep 1; i=$((i+1)); done
|
||||
done
|
||||
wait "$bg"
|
||||
# The duplicate reports `delivered:false` -- truthfully, but about the wrong send.
|
||||
# The first one delivered, so carry that forward, or §1's cleanup reads a completed
|
||||
# turn as an undelivered one and keeps a finished worker forever.
|
||||
r=$(jq -c 'if .success and (.data.wait.ended | not) then .data.delivered = true else . end' < "$tmp")
|
||||
rm -f "$tmp"
|
||||
fi
|
||||
printf '%s\n' "$r"
|
||||
}
|
||||
@@ -247,4 +294,4 @@ last_text() {
|
||||
# The stamp is the LAST line on purpose (a truncated write leaves it unset) and is kept
|
||||
# bare on purpose: the write condition above anchors on it with $, so an inline comment
|
||||
# here would fail that match and rewrite this file on every single bootstrap.
|
||||
CODEMAN_PREAMBLE=1.22.0
|
||||
CODEMAN_PREAMBLE=1.30.1
|
||||
|
||||
@@ -21,7 +21,7 @@ by sourcing the preamble file the §0 bootstrap wrote, and checking its version
|
||||
|
||||
```bash
|
||||
. "${XDG_CACHE_HOME:-$HOME/.cache}/codeman-agent-$CODEMAN_SESSION_ID.sh" 2>/dev/null
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.22.0 ] || { echo "preamble missing or stale; re-run the §0 bootstrap"; exit 1; }
|
||||
[ "${CODEMAN_PREAMBLE:-}" = 1.30.1 ] || { echo "preamble missing or stale; re-run the §0 bootstrap"; exit 1; }
|
||||
```
|
||||
|
||||
Do **not** re-paste the preamble body into each call. Sourcing it is what retires the
|
||||
|
||||
@@ -340,6 +340,35 @@ const capabilitiesSchema = z
|
||||
// an env var, so it declares baseUrl/apiKey injection with no model var at all.
|
||||
modelVars: z.array(envName).max(8),
|
||||
launchModel: launchModelTemplate,
|
||||
// Optional: the env var to carry a discovered per-model context-window size
|
||||
// (claude's CLAUDE_CODE_MAX_CONTEXT_TOKENS), and/or the env var that isolates
|
||||
// this session's config/credential directory from the user's real one (claude's
|
||||
// CLAUDE_CONFIG_DIR) so an injected API key never collides with a stored OAuth
|
||||
// session. See the customModelInjection doc comment in cli-registry/types.ts.
|
||||
contextLengthVar: envName.optional(),
|
||||
configDirVar: envName.optional(),
|
||||
// Relative path, WITHIN the isolated configDirVar directory, of a trust-dialog
|
||||
// seed file the CLI itself owns the shape of — claude's `.claude.json`
|
||||
// `customApiKeyResponses.approved` list, the same field an interactive "Detected
|
||||
// a custom API key — use it?" prompt writes to on a real terminal. Only makes
|
||||
// sense alongside configDirVar (an isolated, otherwise-empty directory has none
|
||||
// of a real profile's prior approvals), and only implemented for the
|
||||
// 'claude-api-key-responses' shape today — see custom-model-injection-apply.ts.
|
||||
apiKeyTrustFile: z
|
||||
.object({ relPath: z.string().min(1).max(80), shape: z.literal('claude-api-key-responses') })
|
||||
.strict()
|
||||
.optional(),
|
||||
// An isolated config directory replays the CLI's whole first-run sequence (theme
|
||||
// picker, security notes, per-project trust dialog, bypass-permissions warning)
|
||||
// on every launch, same root cause as apiKeyTrustFile above — this reuses that
|
||||
// same file to pre-seed the state a real, already-onboarded profile carries. See
|
||||
// the customModelInjection doc comment in cli-registry/types.ts.
|
||||
skipFirstRunPrompts: z.boolean().optional(),
|
||||
// DeepSeek-only, confirmed by reading its own bundled SDK source: it concatenates
|
||||
// "/chat/completions" onto baseUrlVar's value with no "/v1" of its own, while
|
||||
// llama-swap/llama.cpp only serves the "/v1/..." path — claude/gemini must NOT
|
||||
// get this. See the customModelInjection doc comment in cli-registry/types.ts.
|
||||
appendV1Suffix: z.boolean().optional(),
|
||||
})
|
||||
.strict(),
|
||||
z
|
||||
|
||||
@@ -228,15 +228,31 @@ const CLAUDE: CliEntry = {
|
||||
privilegedParams: [],
|
||||
// ANTHROPIC_* is NOT in allowedPrefixes/allowedKeys above (deliberately — see the
|
||||
// allowedPrefixes comment nearby), so these are unreachable via plain envOverrides
|
||||
// today; listed here only so the dedicated custom-model route (docs/custom-model-endpoints-plan.md
|
||||
// chunk 5) clamps them for a non-granted multi-user owner the same way every other
|
||||
// CLI's injection vars are clamped, the day that route widens who can set them.
|
||||
// today. privilegedEnvKeys has exactly one consumer, ownerClampedEnvKeys() in
|
||||
// session-env-clamp.ts, which feeds the generic envOverrides clamp on
|
||||
// POST /api/sessions, POST /api/quick-start and reboot-restore — no custom-model
|
||||
// route reads this field at all, and the values it injects are merged in AFTER
|
||||
// that clamp runs regardless of what's listed here.
|
||||
privilegedEnvKeys: [
|
||||
'ANTHROPIC_BASE_URL',
|
||||
'ANTHROPIC_API_KEY',
|
||||
'ANTHROPIC_DEFAULT_SONNET_MODEL',
|
||||
'ANTHROPIC_DEFAULT_HAIKU_MODEL',
|
||||
'ANTHROPIC_DEFAULT_OPUS_MODEL',
|
||||
// CLAUDE_CODE_MAX_CONTEXT_TOKENS already matches the CLAUDE_CODE_* allowedPrefix, and
|
||||
// CLAUDE_CONFIG_DIR is already an allowed exact key (docs/wiki/Agent-CLIs.md), so both
|
||||
// were already reachable via plain envOverrides before this pair existed and this
|
||||
// feature does not strictly need either listed. They stay listed anyway, because
|
||||
// types.ts's rule ("every traffic-redirecting var this feature introduces MUST also
|
||||
// appear in privilegedEnvKeys") is meant to hold literally, not with an exception
|
||||
// carved out for the two vars that happen not to need it today. The real
|
||||
// consequence lands on the GENERIC envOverrides clamp above, not on this feature:
|
||||
// a non-granted multi-user owner can no longer set CLAUDE_CONFIG_DIR through
|
||||
// envOverrides at all (the per-client-account override, #255), and a PERSISTED one
|
||||
// is now stripped on reboot-restore for such an owner too — see
|
||||
// session-env-clamp.ts's own fileoverview.
|
||||
'CLAUDE_CODE_MAX_CONTEXT_TOKENS',
|
||||
'CLAUDE_CONFIG_DIR',
|
||||
],
|
||||
gates: { nameFlag: { minVersion: '2.1.224', failClosed: true } },
|
||||
// Custom Model Endpoint Profiles (docs/custom-model-endpoints-plan.md) — verified by hand against a real
|
||||
@@ -247,6 +263,32 @@ const CLAUDE: CliEntry = {
|
||||
baseUrlVar: 'ANTHROPIC_BASE_URL',
|
||||
apiKeyVar: 'ANTHROPIC_API_KEY',
|
||||
modelVars: ['ANTHROPIC_DEFAULT_SONNET_MODEL', 'ANTHROPIC_DEFAULT_HAIKU_MODEL', 'ANTHROPIC_DEFAULT_OPUS_MODEL'],
|
||||
// Verified via Claude Code's own docs: CLAUDE_CODE_MAX_CONTEXT_TOKENS overrides the
|
||||
// assumed context window and applies directly for a model name Claude Code doesn't
|
||||
// recognize as one of its own — exactly the custom-model case. Without it, Claude Code
|
||||
// assumes a large (200k) window for any unrecognized model id and never compacts,
|
||||
// eventually overflowing a much smaller real local context (see plan doc reasoning
|
||||
// above the interface for the confirmed failure).
|
||||
contextLengthVar: 'CLAUDE_CODE_MAX_CONTEXT_TOKENS',
|
||||
// Isolates this session's config/credential directory so an injected ANTHROPIC_API_KEY
|
||||
// never shares a directory with a stored claude.ai OAuth login — see the doc comment on
|
||||
// customModelInjection in cli-registry/types.ts for the traded-off side effect.
|
||||
configDirVar: 'CLAUDE_CONFIG_DIR',
|
||||
// ⚠️ Required alongside configDirVar, not optional in practice: verified live that an
|
||||
// isolated, otherwise-empty config directory makes claude stop at an interactive
|
||||
// "Detected a custom API key — use it?" prompt on EVERY launch, defaulting to "No" with
|
||||
// no one at the TTY to answer — silently refusing the very key this feature injected.
|
||||
// Pre-seeding this file's customApiKeyResponses.approved list (verified against a real
|
||||
// ~/.claude.json after answering the prompt once by hand) answers it in advance instead.
|
||||
apiKeyTrustFile: { relPath: '.claude.json', shape: 'claude-api-key-responses' },
|
||||
// ⚠️ Same isolated-directory root cause, one step further: verified live that on top
|
||||
// of the API-key prompt above, a fresh CLAUDE_CONFIG_DIR also replays claude's ENTIRE
|
||||
// first-run sequence on every launch — the theme picker, the security-notes screen,
|
||||
// the per-project "trust this folder?" dialog, and (running with
|
||||
// --dangerously-skip-permissions) a one-time bypass-permissions warning — none of
|
||||
// which a real, already-onboarded profile shows again. Pre-seeds that same
|
||||
// already-onboarded state instead of leaving a human to click through it.
|
||||
skipFirstRunPrompts: true,
|
||||
},
|
||||
},
|
||||
overlays: {
|
||||
@@ -1071,15 +1113,28 @@ const DEEPSEEK: CliEntry = {
|
||||
// privilege rather than granting it, and clamping it here was a real regression
|
||||
// (test/deepseek-mode.test.ts) fixed before this shipped.
|
||||
privilegedEnvKeys: ['DSH_PERMISSION_MODE', 'DSH_HOME', 'DEEPSEEK_BASE_URL'],
|
||||
// Web-researched, unverified, partial: reuses the already-existing DEEPSEEK_BASE_URL/
|
||||
// DEEPSEEK_API_KEY keys above. No modelVars — dsh's model is a profile-composition
|
||||
// entry (see `model: { source: 'none' }` above), not an env var, so forcing a specific
|
||||
// model name may not fully work; verify against a real profile before shipping.
|
||||
// Reuses the already-existing DEEPSEEK_BASE_URL/DEEPSEEK_API_KEY keys above. No
|
||||
// modelVars — dsh's model is a profile-composition entry (see `model: { source: 'none'
|
||||
// }` above), not an env var, so forcing a specific model name may not fully work;
|
||||
// verify against a real profile before shipping.
|
||||
//
|
||||
// ⚠️ appendV1Suffix is REQUIRED, not optional-nice-to-have: without it every request
|
||||
// 404s. Confirmed live and by reading dsh's own bundled source
|
||||
// (@deepseek-ai/dsh-llm-deepseek): it builds the request URL as
|
||||
// `${DEEPSEEK_BASE_URL}/chat/completions` with no "/v1" of its own (its real public
|
||||
// API, https://api.deepseek.com, expects the caller's base URL to already carry any
|
||||
// needed prefix), while llama-swap/llama.cpp only serves the OpenAI-conventional
|
||||
// "/v1/chat/completions" — a bare POST to ".../chat/completions" 404s live, and the
|
||||
// 404 reported here originally ("dsh: HTTP_404: DeepSeek API error (HTTP 404)")
|
||||
// matches dsh's own error-message template for exactly this failure. See the
|
||||
// customModelInjection doc comment in cli-registry/types.ts for the full reasoning,
|
||||
// including why claude/gemini must NOT get this.
|
||||
customModelInjection: {
|
||||
kind: 'env',
|
||||
baseUrlVar: 'DEEPSEEK_BASE_URL',
|
||||
apiKeyVar: 'DEEPSEEK_API_KEY',
|
||||
modelVars: [],
|
||||
appendV1Suffix: true,
|
||||
},
|
||||
},
|
||||
overlays: {
|
||||
|
||||
@@ -496,9 +496,74 @@ export interface CliCapabilities {
|
||||
* declares). Absent = the config alone selects the model (claude's env vars,
|
||||
* opencode's blob, codex's top-level `model` key). Applied by the session's
|
||||
* respawn options through the entry's `legacyConfigField`, never by id.
|
||||
*
|
||||
* `contextLengthVar` (env kind only): the env var a discovered per-model context-window
|
||||
* size is written to when known (claude's `CLAUDE_CODE_MAX_CONTEXT_TOKENS`) — without it,
|
||||
* a CLI that assumes a large default window for an unrecognized model name keeps sending
|
||||
* full-size prompts against a much smaller local server and eventually overflows its real
|
||||
* context (verified: a 33.7K-token system prompt against a 16384-token llama-swap model).
|
||||
* Absent when the CLI has no such override, or the value is unknown for this model.
|
||||
*
|
||||
* `configDirVar` (env kind only): the env var that redirects this session's config/
|
||||
* credential directory to an isolated, per-session one (claude's `CLAUDE_CONFIG_DIR`), so
|
||||
* an injected API key never coexists with a stored claude.ai OAuth session in the same
|
||||
* directory — the CLI still warns "both claude.ai and ANTHROPIC_API_KEY set" when they
|
||||
* share a directory even though the API key wins for actual requests. Isolating it trades
|
||||
* that cosmetic warning for a documented side effect: a relocated config directory writes
|
||||
* transcripts outside `~/.claude/projects`, blinding the response viewer, subagent
|
||||
* windows, and Read My Mind for that session (see docs/wiki/Agent-CLIs.md).
|
||||
*
|
||||
* `apiKeyTrustFile` (env kind only, alongside configDirVar): an isolated config directory
|
||||
* has none of a real profile's prior "detected a custom API key, use it?" approvals, so
|
||||
* without this the CLI stops and asks interactively on every single launch — with no one
|
||||
* at a TTY to answer, that's a hang, not a warning (confirmed live: claude's own default
|
||||
* answer, "No", would silently refuse to use the very key this feature just injected).
|
||||
* `relPath`/`shape` name the file (claude's `.claude.json`) and its
|
||||
* `customApiKeyResponses.approved` field this pre-seeds — the exact field a real answered
|
||||
* prompt itself writes to, so this isn't bypassing the check, just answering it the same
|
||||
* way a one-off prior approval on a shared profile already would.
|
||||
*
|
||||
* `skipFirstRunPrompts` (env kind only, alongside apiKeyTrustFile): an isolated config
|
||||
* directory is not just missing API-key approvals — it is a brand-new profile as far as
|
||||
* the CLI is concerned, so it also replays its ENTIRE first-run sequence on every launch:
|
||||
* the theme picker, the security-notes screen, the per-project "trust this folder?"
|
||||
* dialog, and (running with a bypass-permissions flag) a one-time warning about it —
|
||||
* confirmed live, none of which a real, long-used profile ever shows again. `true`
|
||||
* pre-seeds the same state a real profile accumulates from having answered all of that
|
||||
* once: `hasCompletedOnboarding` and the launching session's own project entry in the
|
||||
* `apiKeyTrustFile` (claude's `.claude.json`), plus `skipDangerousModePermissionPrompt`
|
||||
* in claude's `settings.json` — see `seedFirstRunState`/`seedSkipBypassPermissionsPrompt`
|
||||
* in custom-model-injection-apply.ts. Requires `apiKeyTrustFile` to be set too, since it
|
||||
* reuses that file.
|
||||
*
|
||||
* `appendV1Suffix` (env kind only): the raw `endpoint.baseUrl` gets `withV1Suffix()`
|
||||
* applied before being written to `baseUrlVar`, instead of being used verbatim.
|
||||
* DeepSeek needs this and claude/gemini must NOT get it — a per-CLI asymmetry confirmed
|
||||
* by reading each SDK's own request-building source, not assumed: DeepSeek Harness's
|
||||
* bundled `@deepseek-ai/dsh-llm-deepseek` concatenates `${connection.baseURL}/chat/
|
||||
* completions` with no `/v1` insertion of its own (its real public API base,
|
||||
* `https://api.deepseek.com`, expects the caller's base URL to already carry any
|
||||
* needed prefix), while llama-swap/llama.cpp only ever serves the OpenAI-conventional
|
||||
* `/v1/chat/completions` — confirmed live: a bare `POST <baseUrl>/chat/completions`
|
||||
* 404s, `POST <baseUrl>/v1/chat/completions` succeeds, and the harness's own error
|
||||
* message template (`DeepSeek API error (HTTP ${status})`) reproduces the exact
|
||||
* `HTTP_404` this feature originally shipped with unexplained. Claude Code's own SDK,
|
||||
* by contrast, was already confirmed working end-to-end against the RAW `baseUrl` with
|
||||
* no suffix — appending one there would be wrong, not just redundant.
|
||||
*/
|
||||
customModelInjection:
|
||||
| { kind: 'env'; baseUrlVar: string; apiKeyVar: string; modelVars: string[]; launchModel?: string }
|
||||
| {
|
||||
kind: 'env';
|
||||
baseUrlVar: string;
|
||||
apiKeyVar: string;
|
||||
modelVars: string[];
|
||||
launchModel?: string;
|
||||
contextLengthVar?: string;
|
||||
apiKeyTrustFile?: { relPath: string; shape: 'claude-api-key-responses' };
|
||||
configDirVar?: string;
|
||||
skipFirstRunPrompts?: boolean;
|
||||
appendV1Suffix?: boolean;
|
||||
}
|
||||
| { kind: 'configContentEnv'; envVar: string; template: 'opencode-json'; launchModel?: string }
|
||||
| {
|
||||
kind: 'configDir';
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
/**
|
||||
* @fileoverview Limits shared between Wake-on-LAN parsing and its request schema.
|
||||
*
|
||||
* Its own module because `src/remote-wake.ts` is import-fenced: only
|
||||
* `web/routes/session-routes.ts` and `web/server.ts` may import it, so that no
|
||||
* watcher or boot-recovery path can WAKE a host (pinned by the wiring guard in
|
||||
* `test/remote-wake.test.ts`). `web/schemas.ts` needs the same MAC-count limit and
|
||||
* must not become a third importer, and it would drag `dgram`/`net`/`child_process`
|
||||
* into every module that validates a request body. A plain constant satisfies both.
|
||||
*/
|
||||
|
||||
/**
|
||||
* How many comma-separated MACs one `wakeMac` may carry.
|
||||
*
|
||||
* ⚠ Single source for `parseMacList()` and `RemoteHostSchema.wakeMac`. The two used
|
||||
* to disagree: the schema's 128-character cap admits seven MACs while the parser
|
||||
* rejected more than four all-or-nothing, so a five-MAC value validated, persisted to
|
||||
* `remote-hosts.json`, and then resolved to NO wake target. The host read as
|
||||
* unconfigured and the banner offered "Configure WoL" for a host the user had just
|
||||
* configured, which is the worst shape a validation gap can take: accepted, stored,
|
||||
* silently inert.
|
||||
*/
|
||||
export const MAX_WAKE_MACS = 4;
|
||||
@@ -40,6 +40,38 @@ export interface CustomModelHost {
|
||||
authStyle?: CustomModelAuthStyle;
|
||||
models?: string[];
|
||||
lastDiscoveredAt?: string;
|
||||
/**
|
||||
* The model the Run-menu picker (docs/custom-model-endpoints-plan.md) applies when
|
||||
* this endpoint is picked with no further choice — one generated menu entry per
|
||||
* (CLI, endpoint) pair, not per (CLI, endpoint, model), so it needs a single answer.
|
||||
* Must be a member of `models` when set; the picker falls back to `models[0]` when
|
||||
* this is unset, and disables the entry entirely when `models` is empty (nothing to
|
||||
* default to). Never auto-set on discovery — the previous default staying valid
|
||||
* after a re-discover is a property worth keeping even if the model list changes.
|
||||
*/
|
||||
defaultModelId?: string;
|
||||
/**
|
||||
* Discovered context-window size (tokens) per model id, keyed by the same strings as
|
||||
* `models`. Populated opportunistically during discovery (`custom-model-routes.ts`) from
|
||||
* llama.cpp/llama-swap's `GET /props?model=<id>` — the plain OpenAI-shaped `/v1/models`
|
||||
* response has no such field. Only ever probed for a model the server already reports as
|
||||
* loaded (llama-swap's `status.value === 'loaded'`); an unloaded one is deliberately never
|
||||
* probed, since llama-swap treats `/props?model=` as a routing hint that can trigger an
|
||||
* actual (slow, GPU-swapping) model load as a side effect of merely asking. A model this
|
||||
* has no entry for simply gets no context-length env override applied — never a guess.
|
||||
*/
|
||||
modelContextLengths?: Record<string, number>;
|
||||
/**
|
||||
* Discovered file size (GB) per model id, keyed by the same strings as `models`.
|
||||
* Populated during discovery by parsing llama-swap's own `description` field for an
|
||||
* auto-discovered model ("Auto-discovered 16.35 GB - parameters auto-fitted by
|
||||
* llama.cpp") — a hand-configured profile's own description has no such figure and
|
||||
* correctly gets no entry, never a guess. Used only to label the Run-menu picker's
|
||||
* "loading model" banner with a rough, unmeasured expected-time estimate
|
||||
* (the Run-menu picker's loading banner in session-ui.js) — never a guarantee, and never anything a
|
||||
* server-side check relies on.
|
||||
*/
|
||||
modelSizesGB?: Record<string, number>;
|
||||
}
|
||||
|
||||
export function customModelHostsPath(configDir: string): string {
|
||||
|
||||
@@ -10,7 +10,8 @@
|
||||
* cli-registry changes" requirement it was written against.
|
||||
*/
|
||||
|
||||
import { chmodSync, mkdirSync, writeFileSync, rmSync } from 'node:fs';
|
||||
import { chmodSync, existsSync, mkdirSync, readFileSync, writeFileSync, rmSync, symlinkSync } from 'node:fs';
|
||||
import { homedir, platform } from 'node:os';
|
||||
import { join, dirname } from 'node:path';
|
||||
import { dataPath } from './config/instance.js';
|
||||
import type { CliEntry } from './config/cli-registry/types.js';
|
||||
@@ -48,6 +49,168 @@ export function applyConfigDirInjection(baseDir: string, injection: ConfigDirInj
|
||||
return { [injection.dirEnvVar]: baseDir, ...injection.extraEnv };
|
||||
}
|
||||
|
||||
/**
|
||||
* Real, shared Claude config directory Codeman's own host process runs under — honors
|
||||
* `CLAUDE_CONFIG_DIR` the same way `claude-credentials.ts`'s `claudeCredentialsPath()`
|
||||
* does, so the symlink below points at wherever `~/.claude/projects` actually lives
|
||||
* rather than assuming the plain default.
|
||||
*/
|
||||
function realClaudeConfigDir(): string {
|
||||
const configured = typeof process.env.CLAUDE_CONFIG_DIR === 'string' && process.env.CLAUDE_CONFIG_DIR.trim();
|
||||
return configured || join(homedir(), '.claude');
|
||||
}
|
||||
|
||||
/**
|
||||
* Symlinks `<isolatedDir>/projects` back to the real, shared `~/.claude/projects`, so an
|
||||
* isolated `CLAUDE_CONFIG_DIR` (used to keep an injected API key away from a stored OAuth
|
||||
* session — see `configDirVar` on customModelInjection) doesn't also blind the response
|
||||
* viewer, subagent windows, and Read My Mind for that session (docs/wiki/Agent-CLIs.md).
|
||||
* Best-effort: a platform that refuses symlinks (unprivileged Windows without a junction
|
||||
* fallback working, e.g.) just keeps the pre-existing documented side effect instead of
|
||||
* failing the whole custom-model apply over a nice-to-have.
|
||||
*/
|
||||
function linkSharedProjectsDir(isolatedDir: string): void {
|
||||
const link = join(isolatedDir, 'projects');
|
||||
if (existsSync(link)) return; // already linked (idempotent re-apply) or real dir wrote one
|
||||
try {
|
||||
symlinkSync(join(realClaudeConfigDir(), 'projects'), link, platform() === 'win32' ? 'junction' : 'dir');
|
||||
} catch {
|
||||
// best-effort only — response viewer/subagent windows go blind for this session instead
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Pre-approves the injected API key in an isolated config directory's trust-dialog state
|
||||
* (`customModelInjection.apiKeyTrustFile`), so an otherwise-empty directory doesn't make the
|
||||
* CLI stop at an interactive "Detected a custom API key — use it?" prompt on every single
|
||||
* launch. Confirmed live: with nobody at the TTY to answer, that prompt's own default
|
||||
* ("No") silently refuses the very key this feature just injected — this isn't bypassing
|
||||
* the check, it's answering it the same field a real answered prompt itself writes to
|
||||
* (verified against a real `~/.claude.json` after answering by hand once).
|
||||
*
|
||||
* Merges rather than overwrites: the file may already carry fields the CLI itself wrote on
|
||||
* an earlier launch in this same isolated directory (machineID, userID, other approved
|
||||
* keys), and a corrupt or partially-written file (a crash mid-write) is treated as absent
|
||||
* rather than failing the whole apply over a nice-to-have.
|
||||
*/
|
||||
/**
|
||||
* The form Claude Code actually stores an approved key in: the trimmed last 20
|
||||
* characters. Mirrors the CLI's own `e.trim().slice(-20)`, which is applied on BOTH
|
||||
* the write and the lookup, so anything else never matches.
|
||||
*/
|
||||
export function truncateApiKeyForTrustFile(apiKey: string): string {
|
||||
return apiKey.trim().slice(-20);
|
||||
}
|
||||
|
||||
function seedApiKeyTrustFile(
|
||||
configDir: string,
|
||||
trustFile: { relPath: string; shape: 'claude-api-key-responses' },
|
||||
apiKey: string
|
||||
): void {
|
||||
const filePath = join(configDir, trustFile.relPath);
|
||||
let existing: Record<string, unknown> = {};
|
||||
try {
|
||||
existing = JSON.parse(readFileSync(filePath, 'utf8')) as Record<string, unknown>;
|
||||
} catch {
|
||||
existing = {};
|
||||
}
|
||||
const responses = (existing.customApiKeyResponses ?? {}) as { approved?: unknown; rejected?: unknown };
|
||||
const approved = new Set(Array.isArray(responses.approved) ? (responses.approved as string[]) : []);
|
||||
// ⚠ Claude Code stores and compares only the LAST 20 CHARACTERS of a key, never the
|
||||
// whole thing: its lookup is `approved.includes(key.trim().slice(-20))` (decompiled
|
||||
// from the 2.1.278 bundle, and corroborated by real `~/.claude.json` files, whose
|
||||
// customApiKeyResponses entries are all exactly 20 characters). Seeding the full key
|
||||
// therefore never matches for a REAL key, and claude stops at the interactive
|
||||
// "Detected a custom API key in your environment" prompt, whose default is
|
||||
// "No (recommended)" — so the launch hangs or silently refuses the key this feature
|
||||
// just injected. It went unnoticed because a keyless llama.cpp/llama-swap endpoint
|
||||
// uses DEFAULT_API_KEY ('local-dummy-key', 15 chars), where slice(-20) is the whole
|
||||
// string and the seed matches by accident. Truncating here also keeps a full
|
||||
// third-party credential from being written into a second file on disk.
|
||||
approved.add(truncateApiKeyForTrustFile(apiKey));
|
||||
const rejected = Array.isArray(responses.rejected) ? responses.rejected : [];
|
||||
existing.customApiKeyResponses = { approved: [...approved], rejected };
|
||||
try {
|
||||
writeFileSync(filePath, JSON.stringify(existing, null, 2), { encoding: 'utf8', mode: 0o600 });
|
||||
chmodSync(filePath, 0o600);
|
||||
} catch {
|
||||
// best-effort only — the interactive prompt returns instead of a hard failure here
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Pre-seeds the two remaining pieces of "already been onboarded" state a fresh
|
||||
* `CLAUDE_CONFIG_DIR` has none of (`customModelInjection.skipFirstRunPrompts`, alongside
|
||||
* apiKeyTrustFile): claude replays its whole first-run sequence — the theme picker, the
|
||||
* security-notes screen, and (per-project) the "trust this folder?" dialog — against ANY
|
||||
* config directory that has never completed it, confirmed live against a genuinely fresh
|
||||
* isolated directory. `hasCompletedOnboarding` skips the theme/security-notes screens
|
||||
* outright; `projects[workingDir].hasTrustDialogAccepted` answers the trust dialog for
|
||||
* THIS session's own working directory the same way a real profile's own prior approval
|
||||
* would — other projects in the file are left alone, and `workingDir` is used verbatim
|
||||
* (never realpath'd or slash-normalized) since that's the literal string claude itself
|
||||
* uses as the project key, being whatever string the session was actually launched with
|
||||
* as its cwd.
|
||||
*
|
||||
* Same merge-not-overwrite and corrupt-file-tolerant behavior as `seedApiKeyTrustFile`
|
||||
* (same file, so a second sequential read-modify-write here is deliberate rather than
|
||||
* folding both into one pass — keeps each seed independently testable and optional).
|
||||
*/
|
||||
function seedFirstRunOnboardingState(
|
||||
configDir: string,
|
||||
trustFile: { relPath: string; shape: 'claude-api-key-responses' },
|
||||
workingDir: string
|
||||
): void {
|
||||
const filePath = join(configDir, trustFile.relPath);
|
||||
let existing: Record<string, unknown> = {};
|
||||
try {
|
||||
existing = JSON.parse(readFileSync(filePath, 'utf8')) as Record<string, unknown>;
|
||||
} catch {
|
||||
existing = {};
|
||||
}
|
||||
existing.hasCompletedOnboarding = true;
|
||||
const projects =
|
||||
existing.projects && typeof existing.projects === 'object' && !Array.isArray(existing.projects)
|
||||
? (existing.projects as Record<string, Record<string, unknown>>)
|
||||
: {};
|
||||
const existingProject = projects[workingDir] && typeof projects[workingDir] === 'object' ? projects[workingDir] : {};
|
||||
projects[workingDir] = { ...existingProject, hasTrustDialogAccepted: true };
|
||||
existing.projects = projects;
|
||||
try {
|
||||
writeFileSync(filePath, JSON.stringify(existing, null, 2), { encoding: 'utf8', mode: 0o600 });
|
||||
chmodSync(filePath, 0o600);
|
||||
} catch {
|
||||
// best-effort only — the interactive dialogs return instead of a hard failure here
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Pre-seeds the "skip the bypass-permissions warning" setting (`customModelInjection.
|
||||
* skipFirstRunPrompts`, alongside apiKeyTrustFile) into an isolated config directory's
|
||||
* `settings.json` — a real, already-onboarded profile answers claude's one-time warning
|
||||
* about running with a bypass-permissions flag once and never sees it again, but every
|
||||
* custom-model session launches with a fresh, otherwise-empty CLAUDE_CONFIG_DIR that
|
||||
* carries none of that (confirmed live). A different file from apiKeyTrustFile's
|
||||
* `.claude.json` — this is claude's own global `settings.json`, not project-keyed —
|
||||
* so it gets its own merge-not-overwrite read-modify-write.
|
||||
*/
|
||||
function seedSkipBypassPermissionsPrompt(configDir: string): void {
|
||||
const filePath = join(configDir, 'settings.json');
|
||||
let existing: Record<string, unknown> = {};
|
||||
try {
|
||||
existing = JSON.parse(readFileSync(filePath, 'utf8')) as Record<string, unknown>;
|
||||
} catch {
|
||||
existing = {};
|
||||
}
|
||||
existing.skipDangerousModePermissionPrompt = true;
|
||||
try {
|
||||
writeFileSync(filePath, JSON.stringify(existing, null, 2), { encoding: 'utf8', mode: 0o600 });
|
||||
chmodSync(filePath, 0o600);
|
||||
} catch {
|
||||
// best-effort only — the interactive warning returns instead of a hard failure here
|
||||
}
|
||||
}
|
||||
|
||||
/** Best-effort recursive removal of a previously-written configDir. Never throws. */
|
||||
export function removeConfigDir(dir: string | undefined): void {
|
||||
if (!dir) return;
|
||||
@@ -79,14 +242,43 @@ export function applyCustomModelInjection(
|
||||
entry: Pick<CliEntry, 'capabilities'>,
|
||||
endpoint: CustomModelEndpoint,
|
||||
modelId: string,
|
||||
sessionId: string
|
||||
sessionId: string,
|
||||
/** Discovered context-window size for `modelId`, if known — see `contextLengthVar`. */
|
||||
contextLength?: number,
|
||||
/**
|
||||
* The session's own working directory — only used for `skipFirstRunPrompts`'s per-project
|
||||
* trust-dialog seed, and only when provided (boot recovery, which has no reason to
|
||||
* re-answer a dialog that already fired once, omits it rather than re-deriving it).
|
||||
*/
|
||||
workingDir?: string
|
||||
): AppliedCustomModel | undefined {
|
||||
const injection = buildCustomModelInjection(entry, endpoint, modelId);
|
||||
const injection = buildCustomModelInjection(entry, endpoint, modelId, contextLength);
|
||||
if (injection.kind === 'unsupported') return undefined;
|
||||
if (injection.kind === 'env') {
|
||||
// `configDirVar` (claude's CLAUDE_CONFIG_DIR): point it at the same isolated,
|
||||
// per-session directory the `configDir` kind uses, but write no files into it — an
|
||||
// empty directory has no stored OAuth credential to conflict with the injected API
|
||||
// key, which is the whole point. Reusing the same path keyed by sessionId keeps this
|
||||
// idempotent across a boot-recovery re-apply, same as the configDir kind below.
|
||||
let envOverrides = injection.envOverrides;
|
||||
let configDir: string | undefined;
|
||||
if (injection.configDirVar) {
|
||||
configDir = customModelConfigDir(sessionId);
|
||||
mkdirSync(configDir, { recursive: true, mode: 0o700 });
|
||||
linkSharedProjectsDir(configDir);
|
||||
if (injection.apiKeyTrustFile && injection.apiKey) {
|
||||
seedApiKeyTrustFile(configDir, injection.apiKeyTrustFile, injection.apiKey);
|
||||
}
|
||||
if (injection.skipFirstRunPrompts && injection.apiKeyTrustFile) {
|
||||
if (workingDir) seedFirstRunOnboardingState(configDir, injection.apiKeyTrustFile, workingDir);
|
||||
seedSkipBypassPermissionsPrompt(configDir);
|
||||
}
|
||||
envOverrides = { ...envOverrides, [injection.configDirVar]: configDir };
|
||||
}
|
||||
return {
|
||||
envOverrides: injection.envOverrides,
|
||||
envKeys: Object.keys(injection.envOverrides),
|
||||
envOverrides,
|
||||
envKeys: Object.keys(envOverrides),
|
||||
configDir,
|
||||
launchModel: injection.launchModel,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -16,14 +16,21 @@
|
||||
* shape was rejected by a real codex binary with "invalid type: map,
|
||||
* expected a string" — caught by `scripts/test-local-llm-harnesses.ts`),
|
||||
* but `wire_api = "responses"` is the only value codex still accepts
|
||||
* (support for `"chat"` was dropped in Feb 2026), and a plain OpenAI
|
||||
* Chat-Completions server (llama.cpp, llama-swap, most local setups) does
|
||||
* NOT implement the Responses API — so codex may still fail at the
|
||||
* PROTOCOL level even with a correctly-shaped config file. That gap is
|
||||
* real and current, not a stale warning; see docs/custom-model-endpoints-plan.md. The rest
|
||||
* (gemini/pi/grok/deepseek/omp) have their ONE-SHOT INVOCATION flags
|
||||
* confirmed against real installed binaries' own `--help` output, but
|
||||
* their custom-endpoint env/config conventions remain web-researched,
|
||||
* (support for `"chat"` was dropped in Feb 2026). ⚠️ Re-verified live
|
||||
* against a llama-swap deployment that DOES answer `/v1/responses`: a
|
||||
* plain, no-tool-call turn gets a real reply, but a real tool-call attempt
|
||||
* comes back as `agent_message` TEXT (the tool-call JSON printed as the
|
||||
* answer) rather than a `function_call` item codex would execute —
|
||||
* confirmed via `codex exec --json`'s raw event stream. Tool execution is
|
||||
* what makes codex a coding agent, so this remains not usable for real
|
||||
* work even where plain chat succeeds; see docs/custom-model-endpoints-plan.md
|
||||
* for the full picture (including the harmless `Model metadata ... not
|
||||
* found` warning every custom-endpoint codex session prints — sourced from
|
||||
* a local cache of OpenAI's OWN hosted model catalog that a custom model
|
||||
* can never appear in, confirmed to have no effect on the outcome above).
|
||||
* The rest (gemini/pi/grok/deepseek/omp) have their ONE-SHOT INVOCATION
|
||||
* flags confirmed against real installed binaries' own `--help` output,
|
||||
* but their custom-endpoint env/config conventions remain web-researched,
|
||||
* unverified.
|
||||
*/
|
||||
|
||||
@@ -44,6 +51,20 @@ export interface EnvInjection {
|
||||
envOverrides: Record<string, string>;
|
||||
/** See {@link ConfigDirInjection.launchModel}. */
|
||||
launchModel?: string;
|
||||
/**
|
||||
* Name of the env var the caller should point at an isolated, credential-free config
|
||||
* directory for this session (claude's `CLAUDE_CONFIG_DIR`), from the registry entry's
|
||||
* `customModelInjection.configDirVar`. The actual directory value isn't computed here —
|
||||
* this module is pure and has no sessionId to derive one from — the IO wrapper
|
||||
* (`custom-model-injection-apply.ts`) creates it and adds it to `envOverrides`.
|
||||
*/
|
||||
configDirVar?: string;
|
||||
/** See `customModelInjection.apiKeyTrustFile` — carried through so the IO wrapper can seed it. */
|
||||
apiKeyTrustFile?: { relPath: string; shape: 'claude-api-key-responses' };
|
||||
/** The literal API key value this injection used, for `apiKeyTrustFile` to pre-approve. */
|
||||
apiKey?: string;
|
||||
/** See `customModelInjection.skipFirstRunPrompts` — carried through so the IO wrapper can seed it. */
|
||||
skipFirstRunPrompts?: boolean;
|
||||
}
|
||||
|
||||
export interface ConfigDirInjection {
|
||||
@@ -90,7 +111,9 @@ function quoted(value: string): string {
|
||||
export function buildCustomModelInjection(
|
||||
entry: Pick<CliEntry, 'capabilities'>,
|
||||
endpoint: CustomModelEndpoint,
|
||||
modelId: string
|
||||
modelId: string,
|
||||
/** Discovered context-window size for `modelId`, if known — see `contextLengthVar`. */
|
||||
contextLength?: number
|
||||
): CustomModelInjectionResult {
|
||||
const cap = entry.capabilities.customModelInjection;
|
||||
const apiKey = endpoint.apiKey?.trim() || DEFAULT_API_KEY;
|
||||
@@ -98,11 +121,18 @@ export function buildCustomModelInjection(
|
||||
switch (cap.kind) {
|
||||
case 'env': {
|
||||
const envOverrides: Record<string, string> = {
|
||||
[cap.baseUrlVar]: endpoint.baseUrl,
|
||||
[cap.baseUrlVar]: cap.appendV1Suffix ? withV1Suffix(endpoint.baseUrl) : endpoint.baseUrl,
|
||||
[cap.apiKeyVar]: apiKey,
|
||||
};
|
||||
for (const modelVar of cap.modelVars) envOverrides[modelVar] = modelId;
|
||||
return withLaunchModel({ kind: 'env', envOverrides }, cap.launchModel, modelId);
|
||||
if (cap.contextLengthVar && contextLength !== undefined && Number.isFinite(contextLength)) {
|
||||
envOverrides[cap.contextLengthVar] = String(Math.trunc(contextLength));
|
||||
}
|
||||
let result: EnvInjection = withLaunchModel({ kind: 'env', envOverrides }, cap.launchModel, modelId);
|
||||
if (cap.configDirVar) result = { ...result, configDirVar: cap.configDirVar };
|
||||
if (cap.apiKeyTrustFile) result = { ...result, apiKeyTrustFile: cap.apiKeyTrustFile, apiKey };
|
||||
if (cap.skipFirstRunPrompts) result = { ...result, skipFirstRunPrompts: true };
|
||||
return result;
|
||||
}
|
||||
|
||||
case 'configContentEnv': {
|
||||
@@ -177,11 +207,13 @@ function renderConfigFile(
|
||||
// `env_key`, the NAME of an env var it reads the credential from at runtime, so the
|
||||
// actual value must ride along as an extra env var, never embedded in the file.
|
||||
// ⚠️ `wire_api = "responses"` is the only value codex still accepts (it dropped
|
||||
// `"chat"` support in Feb 2026) — a plain OpenAI Chat-Completions server (llama.cpp,
|
||||
// llama-swap, most local setups) does NOT implement the Responses API, so this
|
||||
// recipe may still fail at the PROTOCOL level even though the file now parses
|
||||
// correctly. That is a real, currently-unresolved compatibility gap, not a syntax
|
||||
// bug — track it before calling codex support done.
|
||||
// `"chat"` support in Feb 2026). Even against a llama-swap deployment that DOES
|
||||
// answer `/v1/responses`, a real tool-call attempt came back as plain TEXT (the
|
||||
// tool-call JSON printed as the model's answer) rather than an executable
|
||||
// `function_call` item — confirmed live via `codex exec --json`. Tool execution is
|
||||
// what makes codex a coding agent, so this remains not usable for real work even
|
||||
// where plain chat succeeds — see the confidence table in
|
||||
// docs/custom-model-endpoints-plan.md, not a syntax bug in this file.
|
||||
const content = [
|
||||
`model = ${quoted(modelId)}`,
|
||||
`model_provider = "custom"`,
|
||||
|
||||
@@ -159,6 +159,15 @@ export interface PaneCaptureOptions {
|
||||
* the 1MB execSync default (ENOBUFS).
|
||||
*/
|
||||
maxCaptureBytes?: number;
|
||||
/**
|
||||
* Filled in by the implementation with the pane geometry the capture was
|
||||
* really taken at, which is not always the geometry the caller last asked
|
||||
* for: a resize and a capture can race, and a pane whose size a desktop
|
||||
* viewport has claimed ignores a smaller client's resize outright. A
|
||||
* visible-frame capture addresses every row absolutely, so a consumer
|
||||
* rendering it needs the real height to know the frame fits.
|
||||
*/
|
||||
capturedGeometry?: { cols: number; rows: number };
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,278 @@
|
||||
/**
|
||||
* @fileoverview Decide which sessions a host reboot destroyed and may be rebuilt.
|
||||
*
|
||||
* A server restart and a host reboot both leave `reconcileSessions()` reporting
|
||||
* dead sessions, and they need opposite handling. A server restart leaves the
|
||||
* tmux panes running, so recovery ATTACHES to them. A host reboot takes the tmux
|
||||
* server down with it, so there is nothing to attach to and the pane has to be
|
||||
* created again. This module holds the decision half of that second case, kept
|
||||
* free of tmux and disk access so it can be unit tested without either. Every
|
||||
* observation it reads is gathered by the caller and passed in.
|
||||
*
|
||||
* "Eligible" here means a session the user did not end on purpose. The rule that
|
||||
* an intentional kill or detach is never auto-revived is enforced at runtime by
|
||||
* an in-memory guard in `TmuxManager`, and memory does not survive a reboot. The
|
||||
* durable equivalent is the record `cleanupSession()` leaves behind. An unpinned
|
||||
* kill deletes the record outright, so it is already absent here. A pinned kill
|
||||
* goes through `demoteOrRemoveSession()` and lands as `status: 'stopped'`, which
|
||||
* is the marker this module refuses. Pruning keeps a pinned record WITHOUT
|
||||
* touching its status, so a pinned session a reboot killed still reads `idle` or
|
||||
* `busy` and stays eligible.
|
||||
*
|
||||
* ⚠️ Ending the AGENT rather than the session is a shape this module CANNOT
|
||||
* recognise today, and a reboot restores it. `/exit` ends the CLI inside the
|
||||
* pane, `remain-on-exit` keeps the pane, and the PTY Codeman owns is the
|
||||
* `tmux attach-session` process, which stays alive throughout — so no exit
|
||||
* handler runs, no lifecycle `exit` is logged, and the record keeps both its pid
|
||||
* and `status: 'idle'`. Nothing durable distinguishes it from a session that was
|
||||
* simply idle when the power went. Ark0N/Codeman#446 covers making Codeman
|
||||
* notice the dead pane; until a record can say the agent is gone, this pass will
|
||||
* offer those sessions back, and the user dismisses or closes them.
|
||||
*
|
||||
* The `pid` check below is therefore NOT that rule. It refuses a record whose
|
||||
* attach process was already gone, which is a session that never started or
|
||||
* whose pane died outright.
|
||||
*
|
||||
* @dependencies types (SessionState), config/cli-registry
|
||||
* @consumedby web/server (plan build at boot), web/routes/reboot-restore-routes
|
||||
*
|
||||
* @module reboot-restore
|
||||
*/
|
||||
|
||||
import type { SessionState } from './types.js';
|
||||
import { getCli } from './config/cli-registry/registry.js';
|
||||
|
||||
/** Session statuses a reboot restore may rebuild. `stopped` is the kill marker. */
|
||||
const RESTORABLE_STATUSES: ReadonlySet<string> = new Set(['idle', 'busy', 'error']);
|
||||
|
||||
/** Observations the reboot heuristic reads. Gathered by the caller, never here. */
|
||||
export interface RebootEvidence {
|
||||
/** Sessions that still had a live pane during reconciliation. */
|
||||
livePaneCount: number;
|
||||
/** Sessions reconciliation just marked dead. */
|
||||
deadSessionCount: number;
|
||||
/** `os.uptime()`, in seconds. */
|
||||
uptimeSeconds: number;
|
||||
/** Newest `lastActivityAt` across the persisted records, in ms since the epoch. */
|
||||
newestPersistedActivityAt: number;
|
||||
/** `Date.now()` when the evidence was gathered, in ms. */
|
||||
now: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Decide whether the machine plausibly rebooted rather than the server restarting.
|
||||
*
|
||||
* Two signals have to agree. The socket must hold no panes at all while state
|
||||
* still lists sessions, which rules out an ordinary server restart. The host
|
||||
* must also have booted after the newest persisted session activity, which is
|
||||
* the corroboration `os.uptime()` provides cheaply. A wiped tmux socket on a
|
||||
* long-uptime host fails the second test, so a user who killed the tmux server
|
||||
* by hand does not get every session offered back to them.
|
||||
*
|
||||
* This heuristic decides whether to ASK, never whether to act. A wrong yes costs
|
||||
* the user a banner they dismiss, because the restore itself waits for a click.
|
||||
*
|
||||
* ⚠️ `os.uptime()` reports the HOST's uptime, which a container shares, and that
|
||||
* cuts BOTH ways rather than simply switching the feature off in Docker. After a
|
||||
* genuine host reboot a containerized Codeman sees the host's short uptime, so the
|
||||
* banner DOES appear and the feature works. What it cannot see is a container-only
|
||||
* restart: the host uptime is long, the boot test fails, and no banner appears
|
||||
* although every in-container pane is gone (`docker/server.Dockerfile` installs
|
||||
* tmux inside the Codeman container, and the self-updater restarts the Compose
|
||||
* deployment by exiting the container, so that is the case where this would help
|
||||
* most). Failing quiet is the safe direction, and closing the gap needs a boot
|
||||
* signal the container owns (PID 1's start time, gated on the existing
|
||||
* `isRunningInContainer()`) rather than a wider heuristic.
|
||||
*/
|
||||
export function looksLikeHostReboot(evidence: RebootEvidence): boolean {
|
||||
if (evidence.deadSessionCount === 0) return false;
|
||||
if (evidence.livePaneCount > 0) return false;
|
||||
if (evidence.newestPersistedActivityAt <= 0) return false;
|
||||
const bootedAt = evidence.now - evidence.uptimeSeconds * 1000;
|
||||
return bootedAt > evidence.newestPersistedActivityAt;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pick the conversation the rebuilt pane should resume.
|
||||
*
|
||||
* The chain's tail is the newest conversation the session was holding, which is
|
||||
* what a compact or a clear leaves behind; `resumeSessionId` covers a session
|
||||
* that was itself started as a resume, and the session id is the original
|
||||
* conversation for everything else.
|
||||
*/
|
||||
export function resolveResumeConversationId(state: SessionState): string {
|
||||
const chain = state.claudeSessionChain;
|
||||
const chainTail = Array.isArray(chain) && chain.length > 0 ? chain[chain.length - 1] : undefined;
|
||||
return chainTail || state.resumeSessionId || state.id;
|
||||
}
|
||||
|
||||
/**
|
||||
* Why one session was passed over. Reported for logging and shown to the user.
|
||||
*
|
||||
* The first seven are decided before anything is built. `capacity-reached` and
|
||||
* `rebuild-failed` can only happen once a click is spending the plan, and they
|
||||
* are the two the banner must not confuse with a missing workspace: one means
|
||||
* "try again after closing something", the other means the CLI would not start.
|
||||
*/
|
||||
export interface RebootRestoreRejection {
|
||||
sessionId: string;
|
||||
reason:
|
||||
| 'no-persisted-record'
|
||||
| 'intentionally-ended'
|
||||
| 'not-running'
|
||||
| 'respawn-blocked'
|
||||
| 'remote-or-docker'
|
||||
| 'unsupported-mode'
|
||||
| 'no-working-dir'
|
||||
| 'workspace-missing'
|
||||
| 'workspace-forbidden'
|
||||
| 'already-live'
|
||||
| 'capacity-reached'
|
||||
| 'rebuild-failed';
|
||||
}
|
||||
|
||||
/** One restorable session, as the banner shows it and the rebuild replays it. */
|
||||
export interface RebootRestoreEntry {
|
||||
sessionId: string;
|
||||
name?: string;
|
||||
workingDir: string;
|
||||
owner?: string;
|
||||
mode: string;
|
||||
/** The conversation the rebuilt pane resumes. */
|
||||
resumeConversationId: string;
|
||||
/**
|
||||
* The persisted record, kept whole so the rebuild can replay what it held.
|
||||
* Read at boot, before pruning deletes it, and held in memory until the click.
|
||||
*/
|
||||
state: SessionState;
|
||||
}
|
||||
|
||||
export interface RebootRestorePlan {
|
||||
restore: RebootRestoreEntry[];
|
||||
skipped: RebootRestoreRejection[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Split the sessions reconciliation just killed into the ones a reboot restore
|
||||
* may offer and the ones it must leave alone.
|
||||
*
|
||||
* @param deadSessionIds Session ids `reconcileSessions()` reported as dead.
|
||||
* @param persisted The `state.json` session records, which `cleanupStaleSessions()`
|
||||
* has not pruned yet at the point this runs.
|
||||
* @param workspaceExists Whether a working directory is still on disk. A tmux
|
||||
* session can outlive its deleted repo, and rebuilding one there would scaffold
|
||||
* an empty tree. The caller owns the disk access; the click re-checks, because
|
||||
* a repo can be deleted between the boot and the click.
|
||||
*/
|
||||
export function planRebootRestore(
|
||||
deadSessionIds: readonly string[],
|
||||
persisted: Readonly<Record<string, SessionState>>,
|
||||
workspaceExists: (workingDir: string) => boolean
|
||||
): RebootRestorePlan {
|
||||
const restore: RebootRestoreEntry[] = [];
|
||||
const skipped: RebootRestoreRejection[] = [];
|
||||
|
||||
for (const sessionId of deadSessionIds) {
|
||||
const state = persisted[sessionId];
|
||||
if (!state) {
|
||||
// An unpinned kill already deleted the record, so absence IS the guard.
|
||||
skipped.push({ sessionId, reason: 'no-persisted-record' });
|
||||
continue;
|
||||
}
|
||||
if (!RESTORABLE_STATUSES.has(state.status)) {
|
||||
// A pinned kill was demoted to `stopped`. Reviving it would undo the kill.
|
||||
skipped.push({ sessionId, reason: 'intentionally-ended' });
|
||||
continue;
|
||||
}
|
||||
if (state.pid === null || state.pid === undefined) {
|
||||
// No attach process when the record was last written: the session never
|
||||
// started, or its pane died outright rather than its agent exiting inside a
|
||||
// surviving pane. Either way there was nothing running to bring back.
|
||||
//
|
||||
// ⚠️ This does NOT catch a session the user ended with `/exit`. See the
|
||||
// module header: that leaves the pid in place, because the pid is the tmux
|
||||
// attach process and `remain-on-exit` keeps it alive.
|
||||
//
|
||||
// Conservative on purpose. A session that somehow persisted no pid while
|
||||
// genuinely running is not offered, and its conversation stays reachable
|
||||
// from the Resume list, which is where every session would be without this
|
||||
// feature.
|
||||
skipped.push({ sessionId, reason: 'not-running' });
|
||||
continue;
|
||||
}
|
||||
if (state.respawnBlocked === true) {
|
||||
// The crash-loop breaker tripped on this pane. Re-creating it restarts the loop.
|
||||
skipped.push({ sessionId, reason: 'respawn-blocked' });
|
||||
continue;
|
||||
}
|
||||
if (state.remote || state.docker) {
|
||||
// Both need another host or a container to be up, which a just-booted machine
|
||||
// cannot promise. The remote reconnect watcher owns the remote case already.
|
||||
skipped.push({ sessionId, reason: 'remote-or-docker' });
|
||||
continue;
|
||||
}
|
||||
// Capability, not a CLI id: this pass resumes by handing the CLI a conversation
|
||||
// id through the top-level `resumeSessionId`, which only a CLI whose history the
|
||||
// claude-jsonl reader understands can consume that way. Others carry their thread
|
||||
// id in their own `<Mode>Config`, which this pass does not thread through.
|
||||
if (getCli(state.mode ?? 'claude')?.capabilities.transcript !== 'claude-jsonl') {
|
||||
skipped.push({ sessionId, reason: 'unsupported-mode' });
|
||||
continue;
|
||||
}
|
||||
if (!state.workingDir) {
|
||||
skipped.push({ sessionId, reason: 'no-working-dir' });
|
||||
continue;
|
||||
}
|
||||
if (!workspaceExists(state.workingDir)) {
|
||||
skipped.push({ sessionId, reason: 'workspace-missing' });
|
||||
continue;
|
||||
}
|
||||
restore.push({
|
||||
sessionId,
|
||||
name: state.name,
|
||||
workingDir: state.workingDir,
|
||||
owner: state.owner,
|
||||
mode: state.mode ?? 'claude',
|
||||
resumeConversationId: resolveResumeConversationId(state),
|
||||
state,
|
||||
});
|
||||
}
|
||||
|
||||
return { restore, skipped };
|
||||
}
|
||||
|
||||
/**
|
||||
* Drop the entries whose conversation is already on screen.
|
||||
*
|
||||
* Hours can pass between the boot that built the plan and the click that spends
|
||||
* it, and the Resume list can reach the same conversation in the meantime. Two
|
||||
* panes running `claude --resume` on one conversation is the failure this
|
||||
* prevents, so a match on either the session id or the conversation id is enough
|
||||
* to skip the entry.
|
||||
*/
|
||||
export function rejectAlreadyLive(
|
||||
entries: readonly RebootRestoreEntry[],
|
||||
liveSessionIds: ReadonlySet<string>,
|
||||
liveConversationIds: ReadonlySet<string>
|
||||
): RebootRestorePlan {
|
||||
const restore: RebootRestoreEntry[] = [];
|
||||
const skipped: RebootRestoreRejection[] = [];
|
||||
for (const entry of entries) {
|
||||
if (liveSessionIds.has(entry.sessionId) || liveConversationIds.has(entry.resumeConversationId)) {
|
||||
skipped.push({ sessionId: entry.sessionId, reason: 'already-live' });
|
||||
continue;
|
||||
}
|
||||
restore.push(entry);
|
||||
}
|
||||
return { restore, skipped };
|
||||
}
|
||||
|
||||
/** Newest `lastActivityAt` across persisted records, or 0 when there are none. */
|
||||
export function newestPersistedActivity(persisted: Readonly<Record<string, SessionState>>): number {
|
||||
let newest = 0;
|
||||
for (const state of Object.values(persisted)) {
|
||||
const stamp = state.lastActivityAt ?? state.createdAt ?? 0;
|
||||
if (stamp > newest) newest = stamp;
|
||||
}
|
||||
return newest;
|
||||
}
|
||||
@@ -540,6 +540,32 @@ export function remoteDisplayPath(
|
||||
return `${remote.username}@${remote.host}:${path}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Refresh HOST-level config on a RESTORED `SessionRemote`.
|
||||
*
|
||||
* A session's `remote` block is persisted at launch time (mux-sessions.json /
|
||||
* state.json) and recovery uses that snapshot, so a field ADDED to the host config
|
||||
* later never reaches an already-running session — not even across a Codeman
|
||||
* restart. That is exactly how a `wakeCommand` added to `remote-hosts.json` would
|
||||
* silently do nothing until the session is relaunched (which for an owned remote
|
||||
* session means killing the remote tmux).
|
||||
*
|
||||
* Deliberately narrow: ONLY `wakeCommand`/`wakeMac` are taken from the host config,
|
||||
* and the host is authoritative for them (removing one in the config turns that
|
||||
* wake path off again). The other host-level fields (`commands`, ssh options) stay as
|
||||
* persisted so this cannot silently change how an existing pane connects.
|
||||
*/
|
||||
export function rehydrateRemoteHostFields<T extends { hostId: string; wakeCommand?: string; wakeMac?: string }>(
|
||||
remote: T | undefined,
|
||||
hostsById: ReadonlyMap<string, RemoteHost>
|
||||
): T | undefined {
|
||||
if (!remote) return remote;
|
||||
const host = hostsById.get(remote.hostId);
|
||||
if (!host) return remote;
|
||||
if (remote.wakeCommand === host.wakeCommand && remote.wakeMac === host.wakeMac) return remote;
|
||||
return { ...remote, wakeCommand: host.wakeCommand, wakeMac: host.wakeMac };
|
||||
}
|
||||
|
||||
export function toSessionRemote(host: RemoteHost, remoteCase: RemoteCase): SessionRemote {
|
||||
return {
|
||||
hostId: host.id,
|
||||
@@ -549,6 +575,10 @@ export function toSessionRemote(host: RemoteHost, remoteCase: RemoteCase): Sessi
|
||||
port: host.port,
|
||||
remotePath: remoteCase.remotePath,
|
||||
commands: host.commands,
|
||||
// Wake-on-LAN command/MAC travel with the session so the input route can wake a
|
||||
// sleeping host without a second config read (see remote-wake.ts).
|
||||
wakeCommand: host.wakeCommand,
|
||||
wakeMac: host.wakeMac,
|
||||
// COD-105 — the COD-104 launch path creates the remote session, so we own it
|
||||
// (an explicit kill may propagate a remote kill-session). Discovered+attached
|
||||
// sessions go through `toAttachedSessionRemote` with `owned: false`.
|
||||
@@ -587,6 +617,10 @@ export function toAttachedSessionRemote(
|
||||
port: host.port,
|
||||
remotePath,
|
||||
commands: host.commands,
|
||||
// An attached session can be woken exactly the same way — the identity of the
|
||||
// creator does not change whether the host is asleep.
|
||||
wakeCommand: host.wakeCommand,
|
||||
wakeMac: host.wakeMac,
|
||||
// Discovered + attached — another Codeman created it. Detach-not-kill.
|
||||
owned: false,
|
||||
remoteSessionName,
|
||||
|
||||
+1035
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,347 @@
|
||||
/**
|
||||
* @fileoverview Automatic session names from the first prompt.
|
||||
*
|
||||
* A new tab is born as `w3-myapp`, which says where it runs and nothing about
|
||||
* what it is doing. Once the user submits a real prompt the tab can carry a
|
||||
* title derived from it (`w3-myapp: fix the login redirect`), and this module
|
||||
* holds the three pure pieces of that: a tracker that reconstructs the composer
|
||||
* text from the keystrokes Codeman forwards, the title heuristic, and the
|
||||
* prefix-preserving composition.
|
||||
*
|
||||
* Deliberately no LLM: the prompt already passes through the input boundary,
|
||||
* so a local title is private, deterministic and identical for every CLI.
|
||||
*
|
||||
* ⚠️ The tracker sits on the raw keystroke stream, which carries far more than
|
||||
* the prompt: cursor keys, mouse reports Codeman forwards to the CLI, bracketed
|
||||
* pastes, Alt chords, the bare Esc that interrupts a turn. Every one of those
|
||||
* once named a tab something wrong (a lone Esc ate the next prompt's first
|
||||
* character; a wheel tick mid-word dropped the first half of the prompt), so
|
||||
* the rules below are explicit per key. The model is a best-effort transcript:
|
||||
* keys whose effect on the composer is knowable are mirrored, keys that leave
|
||||
* the text alone are ignored, and keys that replace it with something the
|
||||
* tracker cannot see (history recall) TAINT the draft so that Enter submits
|
||||
* nothing rather than a fragment. A prompt that yields no title leaves the
|
||||
* session eligible for the next one.
|
||||
*
|
||||
* Only user-originated input is fed here; the Session decides that. Ralph
|
||||
* kick-starts, respawn `/clear`s, cron launches and approval answers all go
|
||||
* through the same write paths and must never become a tab title.
|
||||
*
|
||||
* @module session-auto-name
|
||||
*/
|
||||
|
||||
import { MAX_SESSION_NAME_LENGTH } from './config/terminal-limits.js';
|
||||
|
||||
/**
|
||||
* Longest composer draft kept, in code points. The title is cut from the HEAD
|
||||
* of the prompt, so once the cap is reached further text is counted rather
|
||||
* than kept (backspaces consume that count first). Keeping the tail instead
|
||||
* would turn a long paste into a title made of its last line.
|
||||
*/
|
||||
const MAX_PROMPT_BUFFER_CODE_POINTS = 8_192;
|
||||
|
||||
/** Longest escape sequence collected before the tracker gives up on it. */
|
||||
const MAX_ESCAPE_SEQUENCE_LENGTH = 64;
|
||||
|
||||
/** Longest title, in code points, before it is cut with an ellipsis. */
|
||||
const MAX_AUTO_NAME_CODE_POINTS = 72;
|
||||
|
||||
/**
|
||||
* A sentence boundary is only honoured this far into the prompt, or "e.g. fix
|
||||
* this now" becomes "e.g." and "Ok. Fix the bug" becomes "Ok". Short enough
|
||||
* that a CJK sentence (a dozen code points is a full request) still cuts.
|
||||
*/
|
||||
const MIN_SENTENCE_CODE_POINTS = 8;
|
||||
|
||||
/** A CSI sequence ends at its first byte in this range. */
|
||||
const CSI_FINAL_BYTE = /[\x40-\x7e]/;
|
||||
/** CSI parameter and intermediate bytes; anything else mid-sequence is malformed. */
|
||||
const CSI_BODY_BYTE = /[\x20-\x3f]/;
|
||||
|
||||
/**
|
||||
* `/clear`, `/model opus`, `/ralph-loop:ralph-loop`: a slash followed by a
|
||||
* command word and then whitespace or the end. A path (`/home/me/notes.txt
|
||||
* what is this`) has a second slash where the whitespace should be and so is a
|
||||
* prompt.
|
||||
*/
|
||||
const SLASH_COMMAND_PATTERN = /^\/[a-z][a-z0-9_:-]*(?:\s|$)/i;
|
||||
|
||||
// eslint-disable-next-line no-control-regex
|
||||
const CSI_SEQUENCE_PATTERN = /\x1b\[[\x30-\x3f]*[\x20-\x2f]*[\x40-\x7e]/g;
|
||||
// eslint-disable-next-line no-control-regex
|
||||
const CONTROL_CHAR_PATTERN = /[\x00-\x1f\x7f]/g;
|
||||
const SENTENCE_TERMINATORS = new Set(['.', '!', '?', '。', '!', '?']);
|
||||
|
||||
/**
|
||||
* Reconstructs the composer draft from forwarded keystrokes and reports each
|
||||
* submitted prompt. Input arrives in arbitrary chunks (one keystroke, a paste,
|
||||
* an agent's whole prompt plus Enter), so all state lives across calls.
|
||||
*/
|
||||
export class SubmittedPromptTracker {
|
||||
private buffer = '';
|
||||
private bufferCodePoints = 0;
|
||||
/** Code points typed past the cap; backspaces eat these before real text. */
|
||||
private overflow = 0;
|
||||
/** Escape sequence in progress; a lone ESC means "just saw ESC". */
|
||||
private sequence = '';
|
||||
private inPaste = false;
|
||||
/** The composer holds text the tracker never saw (history recall); Enter submits nothing. */
|
||||
private tainted = false;
|
||||
|
||||
feed(data: string): string[] {
|
||||
const submitted: string[] = [];
|
||||
for (const ch of data) {
|
||||
if (this.sequence) {
|
||||
this.continueSequence(ch);
|
||||
continue;
|
||||
}
|
||||
if (ch === '\x1b') {
|
||||
this.sequence = ch;
|
||||
continue;
|
||||
}
|
||||
this.handleKey(ch, submitted);
|
||||
}
|
||||
// A chunk that ENDS in a lone ESC is the Esc key, not the start of a
|
||||
// sequence: xterm hands each key's whole sequence to one write, and the
|
||||
// programmatic senders (an approval deny sends exactly `\x1b`) send it
|
||||
// alone. Leaving it pending would make the next prompt's first character
|
||||
// look like an Alt chord and swallow it.
|
||||
if (this.sequence === '\x1b') this.sequence = '';
|
||||
return submitted;
|
||||
}
|
||||
|
||||
private continueSequence(ch: string): void {
|
||||
if (this.sequence === '\x1b') {
|
||||
if (ch === '[' || ch === 'O' || ch === ']' || ch === 'P') {
|
||||
this.sequence += ch;
|
||||
return;
|
||||
}
|
||||
this.sequence = ch === '\x1b' ? ch : '';
|
||||
// Alt+Enter inserts a newline in the composer; every other Alt chord
|
||||
// (word movement, Alt+B/F) leaves the text alone.
|
||||
if (ch === '\r' || ch === '\n') this.appendSeparator();
|
||||
return;
|
||||
}
|
||||
|
||||
this.sequence += ch;
|
||||
if (this.sequence.length > MAX_ESCAPE_SEQUENCE_LENGTH) {
|
||||
// Not a sequence any terminal sends; what follows is unknowable, so the
|
||||
// draft is tainted rather than titled after the tail of the garbage.
|
||||
this.sequence = '';
|
||||
this.tainted = true;
|
||||
return;
|
||||
}
|
||||
|
||||
const kind = this.sequence[1];
|
||||
if (kind === '[') {
|
||||
if (CSI_FINAL_BYTE.test(ch)) {
|
||||
const sequence = this.sequence;
|
||||
this.sequence = '';
|
||||
this.handleCsi(sequence);
|
||||
} else if (!CSI_BODY_BYTE.test(ch)) {
|
||||
// Malformed (an ESC [ followed by text): drop the sequence and let the
|
||||
// character count as typed rather than swallowing up to 64 of them.
|
||||
this.sequence = '';
|
||||
this.handleKeyOrEscape(ch);
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (kind === 'O') {
|
||||
// SS3 carries exactly one byte (application-mode cursor keys).
|
||||
this.sequence = '';
|
||||
if (ch === 'A' || ch === 'B') this.tainted = true;
|
||||
return;
|
||||
}
|
||||
// OSC / DCS run to BEL or ST (ESC \).
|
||||
if (ch === '\x07' || this.sequence.endsWith('\x1b\\')) this.sequence = '';
|
||||
}
|
||||
|
||||
private handleKeyOrEscape(ch: string): void {
|
||||
if (ch === '\x1b') {
|
||||
this.sequence = ch;
|
||||
return;
|
||||
}
|
||||
// Only reached mid-chunk from a malformed sequence, where no submission can
|
||||
// be reported; a stray Enter there resets the draft like any other Enter.
|
||||
this.handleKey(ch, []);
|
||||
}
|
||||
|
||||
private handleCsi(sequence: string): void {
|
||||
if (sequence === '\x1b[200~') {
|
||||
this.inPaste = true;
|
||||
return;
|
||||
}
|
||||
if (sequence === '\x1b[201~') {
|
||||
this.inPaste = false;
|
||||
return;
|
||||
}
|
||||
const final = sequence[sequence.length - 1];
|
||||
// Up/Down (with or without modifiers) recall history: the composer now
|
||||
// holds a line this tracker never saw. Everything else leaves the text as
|
||||
// it is: Left/Right/Home/End, Delete (`3~`), Shift+Tab (`Z`), SGR mouse
|
||||
// reports (`<…M`/`m`, forwarded on every wheel tick), focus reports.
|
||||
if (final === 'A' || final === 'B') this.tainted = true;
|
||||
}
|
||||
|
||||
private handleKey(ch: string, submitted: string[]): void {
|
||||
const codePoint = ch.codePointAt(0) ?? 0;
|
||||
if (this.inPaste) {
|
||||
// Pasted newlines are newlines IN the composer, never Enter; they and
|
||||
// the other controls (tabs) become a single separator.
|
||||
if (codePoint < 0x20 || codePoint === 0x7f) this.appendSeparator();
|
||||
else this.append(ch);
|
||||
return;
|
||||
}
|
||||
switch (ch) {
|
||||
case '\r': {
|
||||
const prompt = this.tainted ? '' : this.buffer.trim();
|
||||
if (prompt) submitted.push(prompt);
|
||||
this.reset();
|
||||
return;
|
||||
}
|
||||
case '\n':
|
||||
// Ctrl+J, and the line feed the send-key route injects for Shift+Enter:
|
||||
// a newline inside the composer, so the lines join with a separator.
|
||||
this.appendSeparator();
|
||||
return;
|
||||
case '\x7f':
|
||||
case '\x08':
|
||||
this.backspace();
|
||||
return;
|
||||
case '\x17': // Ctrl+W: word rubout
|
||||
this.killWord();
|
||||
return;
|
||||
case '\x15': // Ctrl+U: line discard
|
||||
case '\x03': // Ctrl+C: clears the composer (or, empty, arms an exit)
|
||||
this.reset();
|
||||
return;
|
||||
case '\x10': // Ctrl+P
|
||||
case '\x0e': // Ctrl+N
|
||||
case '\x12': // Ctrl+R: history search
|
||||
case '\x1f': // Ctrl+_: undo
|
||||
this.tainted = true;
|
||||
return;
|
||||
default:
|
||||
// Tab (the @-mention completer, which only ever extends the token),
|
||||
// cursor chords (Ctrl+A/E/B/F) and the rest of C0 leave the text alone.
|
||||
if (codePoint < 0x20 || codePoint === 0x7f) return;
|
||||
this.append(ch);
|
||||
}
|
||||
}
|
||||
|
||||
/** One space between lines, never a run of them, and none at the start. */
|
||||
private appendSeparator(): void {
|
||||
if (this.overflow > 0) return;
|
||||
if (!this.buffer || /\s$/.test(this.buffer)) return;
|
||||
this.append(' ');
|
||||
}
|
||||
|
||||
private append(ch: string): void {
|
||||
if (this.bufferCodePoints >= MAX_PROMPT_BUFFER_CODE_POINTS) {
|
||||
this.overflow += 1;
|
||||
return;
|
||||
}
|
||||
this.buffer += ch;
|
||||
this.bufferCodePoints += 1;
|
||||
}
|
||||
|
||||
private backspace(): void {
|
||||
if (this.overflow > 0) {
|
||||
this.overflow -= 1;
|
||||
return;
|
||||
}
|
||||
if (!this.buffer) return;
|
||||
const last = this.buffer.charCodeAt(this.buffer.length - 1);
|
||||
const units = last >= 0xdc00 && last <= 0xdfff && this.buffer.length >= 2 ? 2 : 1;
|
||||
this.buffer = this.buffer.slice(0, -units);
|
||||
this.bufferCodePoints -= 1;
|
||||
}
|
||||
|
||||
private killWord(): void {
|
||||
this.overflow = 0;
|
||||
this.buffer = this.buffer.replace(/\S+\s*$/u, '');
|
||||
this.bufferCodePoints = Array.from(this.buffer).length;
|
||||
}
|
||||
|
||||
private reset(): void {
|
||||
this.buffer = '';
|
||||
this.bufferCodePoints = 0;
|
||||
this.overflow = 0;
|
||||
this.tainted = false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Turns a submitted prompt into a title, or null when the prompt is not a task:
|
||||
* empty, a slash command (`/clear`, `/model`), or a `!` shell escape.
|
||||
*/
|
||||
export function deriveAutoSessionName(prompt: string): string | null {
|
||||
const text = prompt.replace(CSI_SEQUENCE_PATTERN, '').replace(CONTROL_CHAR_PATTERN, ' ').replace(/\s+/g, ' ').trim();
|
||||
if (!text || text.startsWith('!') || SLASH_COMMAND_PATTERN.test(text)) return null;
|
||||
return truncateCodePoints(firstSentence(text), MAX_AUTO_NAME_CODE_POINTS);
|
||||
}
|
||||
|
||||
/**
|
||||
* The first sentence, provided it is long enough to be one; a trailing full
|
||||
* stop is dropped because a tab title is not a sentence.
|
||||
*/
|
||||
function firstSentence(text: string): string {
|
||||
const codePoints = Array.from(text);
|
||||
for (let i = MIN_SENTENCE_CODE_POINTS - 1; i < codePoints.length; i++) {
|
||||
if (!SENTENCE_TERMINATORS.has(codePoints[i])) continue;
|
||||
const next = codePoints[i + 1];
|
||||
if (next !== undefined && !/\s/.test(next)) continue;
|
||||
return codePoints
|
||||
.slice(0, i + 1)
|
||||
.join('')
|
||||
.replace(/[.。]+$/, '');
|
||||
}
|
||||
return text.replace(/[.。]+$/, '');
|
||||
}
|
||||
|
||||
/** Cuts to `max` code points with an ellipsis, on a word boundary when one is near the end. */
|
||||
function truncateCodePoints(text: string, max: number): string {
|
||||
const codePoints = Array.from(text);
|
||||
if (codePoints.length <= max) return text;
|
||||
let cut = codePoints.slice(0, max - 1).join('');
|
||||
const lastSpace = cut.lastIndexOf(' ');
|
||||
if (lastSpace >= Math.floor(cut.length / 2)) cut = cut.slice(0, lastSpace);
|
||||
return `${cut.trimEnd()}…`;
|
||||
}
|
||||
|
||||
/**
|
||||
* The name a placeholder becomes: `<prefix>: <title>`, so the tab keeps its
|
||||
* case identity and its `w<n>` counter (the tab strip already renders that
|
||||
* form as the title alone, prefix in the tooltip, and the next-session counter
|
||||
* still matches it). A session with no name at all just takes the title. The
|
||||
* result honours `maxLength` in UTF-16 units, the unit the rename route caps.
|
||||
*/
|
||||
export function composeAutoSessionName(
|
||||
currentName: string,
|
||||
title: string,
|
||||
maxLength = MAX_SESSION_NAME_LENGTH
|
||||
): string {
|
||||
const prefix = currentName.trim();
|
||||
if (!prefix) return fitTitle(title, maxLength);
|
||||
const room = maxLength - prefix.length - 2;
|
||||
if (room <= 0) return prefix;
|
||||
return `${prefix}: ${fitTitle(title, room)}`;
|
||||
}
|
||||
|
||||
/** Fits a title into `maxUnits` UTF-16 units, ellipsis included. */
|
||||
function fitTitle(title: string, maxUnits: number): string {
|
||||
if (title.length <= maxUnits) return title;
|
||||
let units = 0;
|
||||
let keep = 0;
|
||||
for (const codePoint of Array.from(title)) {
|
||||
if (units + codePoint.length > maxUnits - 1) break;
|
||||
units += codePoint.length;
|
||||
keep += 1;
|
||||
}
|
||||
return truncateCodePoints(title, keep + 1);
|
||||
}
|
||||
|
||||
/** Codeman's own `w<n>-<case>` / `s<n>-<case>` placeholders, the only names auto-naming replaces. */
|
||||
export function isGeneratedSessionName(name: string): boolean {
|
||||
return /^[ws]\d+-[a-zA-Z0-9_-]+$/.test(name);
|
||||
}
|
||||
@@ -0,0 +1,102 @@
|
||||
/**
|
||||
* @fileoverview The env-var half of the multi-user privilege clamp.
|
||||
*
|
||||
* A session's `envOverrides` can hand back privilege that the per-CLI config
|
||||
* clamp removed, so a non-granted owner's overrides get the privileged keys
|
||||
* stripped before the session is built. The create and resume routes are what
|
||||
* this bites on: they clamp what a request asked for.
|
||||
*
|
||||
* The reboot-restore route calls it as defence in depth, and it CAN strip
|
||||
* something today: `Session.getEnvOverridesForPersist()` keeps only
|
||||
* `CLAUDE_CODE_*` and `CLAUDE_CONFIG_DIR` out of a session's overrides, and
|
||||
* claude's `privilegedEnvKeys` now includes both `CLAUDE_CODE_MAX_CONTEXT_TOKENS`
|
||||
* and `CLAUDE_CONFIG_DIR` (Custom Model Endpoint Profiles, since both can
|
||||
* redirect a claude session's traffic — see stock.ts's own comment on why they
|
||||
* are listed despite not needing the clamp for that feature). So a non-granted
|
||||
* owner's persisted `CLAUDE_CONFIG_DIR` (the per-client-account override, #255)
|
||||
* is now stripped on reboot-restore, silently returning that session to the
|
||||
* default Claude account rather than the account it was pointed at. The grant
|
||||
* re-resolution that ALSO bites on that path is `resolveClaudeModeForUsername`,
|
||||
* which recomputes the permission mode.
|
||||
*
|
||||
* This lives outside `web/routes` on purpose. The question it answers is about
|
||||
* session privilege rather than about HTTP, and `cron/cron-service.ts` sets the
|
||||
* precedent by importing `canUsernameRunPrivilegedCommands` from `user-store.ts`
|
||||
* directly and re-resolving the owner's grant when a job fires. Every caller here
|
||||
* re-resolves the grant at the moment it builds a session, for the same reason.
|
||||
*
|
||||
* @dependencies user-store (canUsernameRunPrivilegedCommands), config/cli-registry
|
||||
* @consumedby web/routes/session-routes, web/routes/reboot-restore-routes
|
||||
*
|
||||
* @module session-env-clamp
|
||||
*/
|
||||
|
||||
import { canUsernameRunPrivilegedCommands } from './user-store.js';
|
||||
import { enabledClis } from './config/cli-registry/registry.js';
|
||||
|
||||
/**
|
||||
* Env-var keys a non-granted owner must not be able to set, because each one
|
||||
* hands back privilege `clampExternalCliBypassForOwner()` just removed, or redirects a
|
||||
* credential-resolution endpoint.
|
||||
*
|
||||
* The DeepSeek three are reachable because `DSH_*` and `DEEPSEEK_*` are
|
||||
* allowlisted `envOverrides` prefixes (schemas.ts) — which they have to be, since
|
||||
* that is also how a user configures the harness's non-privileged knobs.
|
||||
*
|
||||
* - `DSH_PERMISSION_MODE` IS the harness's permission switch. Every other CLI's
|
||||
* bypass is a command-line FLAG, reachable only through the per-CLI config the
|
||||
* clamp already owns; this one is an env var, so the config clamp alone is
|
||||
* half a gate.
|
||||
* - `DSH_HOME` points the launcher at a profile tree, and a profile's plugin code
|
||||
* executes at BOOT, before any approval row can apply. A user who can write a
|
||||
* workspace can put a profile in it, so this is the wider of the two.
|
||||
* - `DEEPSEEK_BASE_URL` aims the provider endpoint, and `_configureCliEnv()`
|
||||
* forwards the SERVER's own `DEEPSEEK_API_KEY` into every dsh pane before
|
||||
* `applyEnvOverrides()` runs — so a non-granted owner who could set the base
|
||||
* URL would have the operator's API key sent as a bearer credential to a host
|
||||
* of their choosing. (`DEEPSEEK_API_KEY` itself stays overridable: supplying
|
||||
* your OWN key removes privilege rather than granting it.)
|
||||
* - `OMP_AUTH_BROKER_URL`/`OMP_AUTH_BROKER_TOKEN` are where omp resolves
|
||||
* credentials from — the same shape as `DEEPSEEK_BASE_URL` above, reachable
|
||||
* because `OMP_*` is an allowlisted prefix. Unlike DeepSeek, Codeman does not
|
||||
* forward any operator-held key into an omp pane today (omp's provider
|
||||
* credentials live in `~/.omp` config files, not env vars), so there is no
|
||||
* known concrete exfiltration path yet — clamped defensively anyway, since a
|
||||
* non-granted owner redirecting where a shared multi-tenant deployment
|
||||
* resolves auth from is not something to allow silently (found in
|
||||
* Ark0N/Codeman#353 review; omp's own knobs are otherwise mostly `PI_*`,
|
||||
* already allowlisted for pi and not addressed here — see resolveOmpHome()).
|
||||
*/
|
||||
export function ownerClampedEnvKeys(): string[] {
|
||||
return enabledClis().flatMap((entry) => entry.capabilities.privilegedEnvKeys);
|
||||
}
|
||||
|
||||
/**
|
||||
* Env-var half of the multi-user bypass clamp.
|
||||
*
|
||||
* `clampExternalCliBypassForOwner()` in `web/routes/session-routes.ts` clamps the
|
||||
* per-CLI CONFIG, and for every CLI
|
||||
* but DeepSeek that is the whole story. Here it is not: `applyEnvOverrides()` runs
|
||||
* AFTER `_configureCliEnv()` in tmux-manager, so an override sent on the SAME
|
||||
* request lands last and wins, and a non-granted owner could restore
|
||||
* `danger-full-access` on the very request the config clamp downgraded.
|
||||
*
|
||||
* Keys are DROPPED rather than rewritten: dropping falls through to what
|
||||
* `_configureCliEnv()` exports, which is the clamped config and the server's own
|
||||
* `DSH_HOME`, i.e. exactly the intended state. No-op in single-user mode and for a
|
||||
* granted owner, like every other clamp here
|
||||
* (`canUsernameRunPrivilegedCommands()` returns true when `!isMultiUserMode()`),
|
||||
* and it returns the caller's own object untouched when there is nothing to strip.
|
||||
*/
|
||||
export async function clampEnvOverridesForOwner(
|
||||
owner: string | undefined,
|
||||
envOverrides: Record<string, string> | undefined
|
||||
): Promise<Record<string, string> | undefined> {
|
||||
if (!envOverrides) return envOverrides;
|
||||
const keys = ownerClampedEnvKeys();
|
||||
if (!keys.some((key) => key in envOverrides)) return envOverrides;
|
||||
if (await canUsernameRunPrivilegedCommands(owner)) return envOverrides;
|
||||
const clamped = { ...envOverrides };
|
||||
for (const key of keys) delete clamped[key];
|
||||
return clamped;
|
||||
}
|
||||
@@ -0,0 +1,133 @@
|
||||
/**
|
||||
* @fileoverview Verify that a programmatically sent prompt actually LEFT the composer,
|
||||
* and press Enter again while it has not.
|
||||
*
|
||||
* Claude Code 2.1.277 (auto-installed 2026-09-18) takes typed text the moment its
|
||||
* composer paints but ignores Enter for the first 30 to 50 seconds after it, so the
|
||||
* `send-keys -l <text>` + `send-keys Enter` pair `TmuxManager.sendInput()` sends 50 ms
|
||||
* apart leaves the prompt sitting on the composer with `0 tokens`, and every caller
|
||||
* that then waits for the turn (send-and-wait, the agent skill, the maintainer bot,
|
||||
* cron, Ralph) burns its whole timeout on a turn that never started. Measured through
|
||||
* the input route on 2026-09-19: an Enter at 28 s stranded, one at 51 s submitted.
|
||||
*
|
||||
* The rule: after a write that carried a carriage return, read the pane on a short
|
||||
* schedule; while the LAST composer line (the CLI's own prompt glyph) still holds the
|
||||
* head of what was sent, send Enter again. An empty composer ends it, and so does a
|
||||
* composer holding anything else, because that text is the user's or the CLI's, never
|
||||
* ours. A pane with no composer line at all (a shell, a CLI whose glyph is not
|
||||
* declared, a direct-PTY session with no pane to read) does nothing: this runs for
|
||||
* EVERY programmatic sender, so a blind Enter here could confirm a dialog nobody asked
|
||||
* about. The composer is the last glyph line on purpose: Claude Code echoes a submitted
|
||||
* prompt with the same glyph higher up in the transcript, so only the last one says
|
||||
* whether the text was taken.
|
||||
*
|
||||
* Pure apart from the injected capture, send and log, so the schedule, the cap and
|
||||
* every stop condition are unit-tested with fake timers (test/session-submit-verifier.test.ts).
|
||||
*/
|
||||
import { stripAnsi } from './utils/index.js';
|
||||
|
||||
/**
|
||||
* When to look, counted from the write: 2 s catches the common case (taken) with one
|
||||
* capture, and the tail reaches 60 s, past twice the longest window measured. Enter is
|
||||
* re-sent at every check that still finds the prompt, so a 50 s window costs about
|
||||
* seven Enters and one capture each; a taken prompt costs one capture.
|
||||
*/
|
||||
export const SUBMIT_VERIFY_DELAYS_MS: readonly number[] = [
|
||||
2_000, 3_000, 5_000, 5_000, 5_000, 10_000, 10_000, 10_000, 10_000,
|
||||
];
|
||||
|
||||
/** How many leading characters of the prompt have to match, whitespace removed. */
|
||||
const PROMPT_HEAD_CHARS = 24;
|
||||
|
||||
const compact = (s: string): string => s.replace(/\s+/g, '');
|
||||
|
||||
/**
|
||||
* Whether `prompt` is still sitting unsubmitted in the composer of `screen`.
|
||||
*
|
||||
* - `true`: the last `glyph` line holds the prompt's head.
|
||||
* - `false`: the composer is empty (the prompt was taken) or holds other text.
|
||||
* - `undefined`: no composer line at all; nothing can be said, so nothing is sent.
|
||||
*
|
||||
* Whitespace is removed on both sides before comparing, because the composer wraps a
|
||||
* long prompt onto indented continuation lines and Claude Code draws a no-break space
|
||||
* after the glyph; `\s` covers that one in JavaScript.
|
||||
*/
|
||||
export function promptStillInComposer(screen: string, prompt: string, glyph: string): boolean | undefined {
|
||||
if (!glyph) return undefined;
|
||||
const composerLines = stripAnsi(screen)
|
||||
.split('\n')
|
||||
.map((l) => l.trim())
|
||||
.filter((l) => l.startsWith(glyph));
|
||||
if (composerLines.length === 0) return undefined;
|
||||
const composer = compact(composerLines[composerLines.length - 1].slice(glyph.length));
|
||||
if (!composer) return false;
|
||||
const head = compact(prompt).slice(0, PROMPT_HEAD_CHARS);
|
||||
return head.length > 0 && composer.startsWith(head);
|
||||
}
|
||||
|
||||
export interface SubmitVerifierDeps {
|
||||
/** The rendered pane, or null when there is none to read. */
|
||||
capture: () => string | null | undefined;
|
||||
/** Press Enter once. Failures are swallowed; the next check decides again. */
|
||||
sendEnter: () => Promise<unknown> | unknown;
|
||||
/** The CLI's composer glyph, resolved at check time (the registry can change). */
|
||||
glyph: () => string;
|
||||
log?: (message: string) => void;
|
||||
/** Test seam; production uses SUBMIT_VERIFY_DELAYS_MS. */
|
||||
delaysMs?: readonly number[];
|
||||
}
|
||||
|
||||
/**
|
||||
* One per session. `arm(text)` starts the schedule for the prompt just sent and
|
||||
* cancels any earlier one: a newer write owns the composer now, and re-sending Enter
|
||||
* for an older prompt could submit the newer one early. `cancel()` is for teardown.
|
||||
*/
|
||||
export class SubmitVerifier {
|
||||
private timer: NodeJS.Timeout | null = null;
|
||||
private generation = 0;
|
||||
|
||||
constructor(private readonly deps: SubmitVerifierDeps) {}
|
||||
|
||||
arm(text: string): void {
|
||||
this.cancel();
|
||||
const gen = this.generation;
|
||||
const delays = this.deps.delaysMs ?? SUBMIT_VERIFY_DELAYS_MS;
|
||||
let step = 0;
|
||||
let elapsed = 0;
|
||||
let resent = 0;
|
||||
|
||||
const schedule = (): void => {
|
||||
if (step >= delays.length) return;
|
||||
const delay = delays[step++];
|
||||
elapsed += delay;
|
||||
this.timer = setTimeout(() => void check(), delay);
|
||||
this.timer.unref?.();
|
||||
};
|
||||
const check = async (): Promise<void> => {
|
||||
this.timer = null;
|
||||
if (gen !== this.generation) return;
|
||||
const screen = this.deps.capture();
|
||||
if (promptStillInComposer(screen ?? '', text, this.deps.glyph()) !== true) return;
|
||||
resent++;
|
||||
this.deps.log?.(
|
||||
`prompt still in the composer after ${Math.round(elapsed / 1000)}s, re-sending Enter (${resent}/${delays.length})`
|
||||
);
|
||||
try {
|
||||
await this.deps.sendEnter();
|
||||
} catch {
|
||||
// The next check re-reads the screen and decides again.
|
||||
}
|
||||
if (gen !== this.generation) return;
|
||||
schedule();
|
||||
};
|
||||
schedule();
|
||||
}
|
||||
|
||||
cancel(): void {
|
||||
this.generation++;
|
||||
if (this.timer) {
|
||||
clearTimeout(this.timer);
|
||||
this.timer = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
+136
-7
@@ -23,7 +23,7 @@
|
||||
* ralph-tracker (todo/completion parsing), bash-tool-parser (tool invocation tracking),
|
||||
* task-tracker (background tasks), mux-interface (tmux abstraction)
|
||||
* @consumedby session-manager, web/server, respawn-controller
|
||||
* @emits session:terminal, session:idle, session:working, session:completion, session:exit
|
||||
* @emits session:terminal, session:idle, session:working, session:completion, session:promptSubmitted, session:exit
|
||||
*
|
||||
* @module session
|
||||
*/
|
||||
@@ -58,6 +58,8 @@ import {
|
||||
type OmpConfig,
|
||||
type SessionRemote,
|
||||
type SessionDocker,
|
||||
type SessionNameSource,
|
||||
type SessionWriteOptions,
|
||||
} from './types.js';
|
||||
import { resolveAndClaimOmpSessionId } from './utils/omp-session-resolver.js';
|
||||
import { probeDockerCliVersion } from './docker-hosts.js';
|
||||
@@ -108,6 +110,7 @@ import {
|
||||
import { DEFAULT_TMUX_HISTORY_LIMIT } from './config/terminal-history.js';
|
||||
import { EXEC_TIMEOUT_MS } from './config/exec-timeout.js';
|
||||
import { getCli } from './config/cli-registry/registry.js';
|
||||
import { SubmitVerifier } from './session-submit-verifier.js';
|
||||
import { compileVersionRegex } from './config/cli-registry/patterns.js';
|
||||
import { resolveSessionCliVersion } from './utils/cli-resolver.js';
|
||||
import {
|
||||
@@ -121,6 +124,7 @@ import { SessionAutoOps } from './session-auto-ops.js';
|
||||
import { detectUsageLimitPause } from './usage-limit-patterns.js';
|
||||
import { SessionTaskCache } from './session-task-cache.js';
|
||||
import { InteractivePtyExitBreaker } from './session-pty-exit-breaker.js';
|
||||
import { isGeneratedSessionName, SubmittedPromptTracker } from './session-auto-name.js';
|
||||
import { parseTerminalAttachmentRequests } from './attachment-magic.js';
|
||||
import {
|
||||
sanitizeAttachmentHistory,
|
||||
@@ -423,6 +427,15 @@ export class Session extends EventEmitter {
|
||||
private _taskCache = new SessionTaskCache();
|
||||
|
||||
private _name: string;
|
||||
private _nameSource: SessionNameSource;
|
||||
/**
|
||||
* Reconstructs the composer draft from USER keystrokes so the first real
|
||||
* prompt can name the tab. Fed only when a write says `fromUser`, and never
|
||||
* for a CLI whose Enter runs a command rather than submitting a prompt
|
||||
* (`startMode: 'shell'`), so a shell tab is not renamed after every `ls`.
|
||||
*/
|
||||
private readonly _submittedPromptTracker = new SubmittedPromptTracker();
|
||||
private readonly _acceptsPrompts: boolean;
|
||||
private ptyProcess: pty.IPty | null = null;
|
||||
private _pid: number | null = null;
|
||||
private _status: SessionStatus = 'idle';
|
||||
@@ -490,6 +503,8 @@ export class Session extends EventEmitter {
|
||||
private _trustDialogAttempts = 0; // Keystrokes sent at the trust dialog
|
||||
private _lastTrustDialogScanAt = 0; // Throttle for the trust-dialog screen read
|
||||
private _trustDialogTimer: NodeJS.Timeout | null = null; // Re-read after a keystroke (see below)
|
||||
/** Re-sends Enter while a programmatic prompt still sits in the composer (session-submit-verifier.ts). */
|
||||
private _submitVerifier: SubmitVerifier | null = null;
|
||||
private _interactiveStartedAt = 0; // When the interactive pane launched (bounds that scan)
|
||||
private _taskTracker: TaskTracker;
|
||||
|
||||
@@ -654,6 +669,12 @@ export class Session extends EventEmitter {
|
||||
workingDir: string;
|
||||
mode?: SessionMode;
|
||||
name?: string;
|
||||
/**
|
||||
* Who owns the name (see `SessionNameSource`). Omitted, it is inferred
|
||||
* from the name: Codeman's own `w<n>-<case>` placeholders (or no name)
|
||||
* stay eligible for auto-naming, anything else counts as the user's.
|
||||
*/
|
||||
nameSource?: SessionNameSource;
|
||||
/** Terminal multiplexer instance (tmux) */
|
||||
mux?: TerminalMultiplexer;
|
||||
/** Whether to use multiplexer wrapping */
|
||||
@@ -723,6 +744,9 @@ export class Session extends EventEmitter {
|
||||
this.createdAt = config.createdAt || Date.now();
|
||||
this.mode = config.mode || 'claude';
|
||||
this._name = config.name || '';
|
||||
this._nameSource =
|
||||
config.nameSource ?? (!this._name || isGeneratedSessionName(this._name) ? 'placeholder' : 'manual');
|
||||
this._acceptsPrompts = getCli(this.mode)?.capabilities.startMode !== 'shell';
|
||||
this._resumeSessionId = config.resumeSessionId;
|
||||
// NOW, not `createdAt`: recovery passes the ORIGINAL creation time of a
|
||||
// days-old tmux session, and seeding last-activity from it would report a
|
||||
@@ -1370,8 +1394,31 @@ export class Session extends EventEmitter {
|
||||
return this._name;
|
||||
}
|
||||
|
||||
/** An explicit rename: the name is the user's from here on and auto-naming never touches it. */
|
||||
set name(value: string) {
|
||||
this._name = value;
|
||||
this._nameSource = 'manual';
|
||||
}
|
||||
|
||||
/**
|
||||
* Names the tab after its first prompt. Only a placeholder is eligible, and
|
||||
* the session stops being one whether or not the string changed: "first
|
||||
* prompt" means the first, not "every prompt until a rename". Returns
|
||||
* whether the name changed, so the caller knows whether to persist and
|
||||
* broadcast.
|
||||
*/
|
||||
applyAutoName(value: string): boolean {
|
||||
if (this._nameSource !== 'placeholder') return false;
|
||||
const name = value.trim();
|
||||
if (!name) return false;
|
||||
this._nameSource = 'auto';
|
||||
if (this._name === name) return false;
|
||||
this._name = name;
|
||||
return true;
|
||||
}
|
||||
|
||||
get nameSource(): SessionNameSource {
|
||||
return this._nameSource;
|
||||
}
|
||||
|
||||
setAutoClear(enabled: boolean, threshold?: number): void {
|
||||
@@ -1455,6 +1502,19 @@ export class Session extends EventEmitter {
|
||||
this._pinnedAt = pinned ? Date.now() : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Restore a pin from a persisted record, keeping the moment it was pinned.
|
||||
*
|
||||
* `setPinned()` stamps `pinnedAt` with now, which is right for a user pinning a
|
||||
* session and wrong for a restore: the session-manager orders its pinned group
|
||||
* by that stamp, so a restored session would jump to the front of a list it had
|
||||
* been sitting further down.
|
||||
*/
|
||||
restorePin(pinned: boolean, pinnedAt?: number): void {
|
||||
this._pinned = pinned;
|
||||
this._pinnedAt = pinned ? (pinnedAt ?? Date.now()) : null;
|
||||
}
|
||||
|
||||
get flickerFilterEnabled(): boolean {
|
||||
return this._flickerFilterEnabled;
|
||||
}
|
||||
@@ -1515,6 +1575,7 @@ export class Session extends EventEmitter {
|
||||
// attach repaint, so the home screens' quiet ordering survives a restart.
|
||||
lastActivityAt: this._wireActivityAt,
|
||||
name: this._name,
|
||||
nameSource: this._nameSource,
|
||||
mode: this.mode,
|
||||
autoClearEnabled: this._autoOps.autoClearEnabled,
|
||||
autoClearThreshold: this._autoOps.autoClearThreshold,
|
||||
@@ -3132,6 +3193,9 @@ export class Session extends EventEmitter {
|
||||
}
|
||||
|
||||
private _clearAllTimers(): void {
|
||||
// Stop re-sending Enter for a prompt this session will never take now
|
||||
this._submitVerifier?.cancel();
|
||||
this._submitVerifier = null;
|
||||
// Clear the workspace-trust follow-up read
|
||||
if (this._trustDialogTimer) {
|
||||
clearTimeout(this._trustDialogTimer);
|
||||
@@ -3513,10 +3577,11 @@ export class Session extends EventEmitter {
|
||||
* discards the data, but it used to do so with no signal at all — which is how
|
||||
* input could disappear while the caller believed it had been delivered.
|
||||
*/
|
||||
write(data: string): boolean {
|
||||
this._trackSubmit(data);
|
||||
write(data: string, options: SessionWriteOptions = {}): boolean {
|
||||
const submittedPrompt = this._trackSubmit(data, options);
|
||||
if (!this.ptyProcess) return false;
|
||||
this.ptyProcess.write(data);
|
||||
this._emitSubmittedPrompt(submittedPrompt);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -3533,10 +3598,35 @@ export class Session extends EventEmitter {
|
||||
return this._lastSubmitAt;
|
||||
}
|
||||
|
||||
private _trackSubmit(data: string): void {
|
||||
/**
|
||||
* Stamps the pane's last Enter for EVERY write, and feeds the auto-name
|
||||
* tracker only for user-originated input on a prompt-taking CLI. Ralph
|
||||
* kick-starts, respawn `/clear`s, cron launches, approval answers and the
|
||||
* trust-dialog keys all arrive without `fromUser` and so can never name a tab.
|
||||
*/
|
||||
private _trackSubmit(data: string, options: SessionWriteOptions): string[] {
|
||||
const submitted = options.fromUser && this._acceptsPrompts ? this._submittedPromptTracker.feed(data) : [];
|
||||
if (data.includes('\r') || data.includes('\n')) {
|
||||
this._lastSubmitAt = Date.now();
|
||||
}
|
||||
return submitted;
|
||||
}
|
||||
|
||||
/**
|
||||
* Feeds user input that reaches the pane AROUND the write paths: the
|
||||
* send-key route injects Shift+Enter's line feed through `tmux send-keys -H`
|
||||
* directly, and without this the two lines of a prompt joined with no
|
||||
* separator. Reports submissions like a write would (a line feed never is one).
|
||||
*/
|
||||
trackUserInput(data: string): void {
|
||||
if (!this._acceptsPrompts) return;
|
||||
this._emitSubmittedPrompt(this._submittedPromptTracker.feed(data));
|
||||
}
|
||||
|
||||
private _emitSubmittedPrompt(prompts: string[]): void {
|
||||
for (const prompt of prompts) {
|
||||
this.emit('promptSubmitted', prompt);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -3633,19 +3723,58 @@ export class Session extends EventEmitter {
|
||||
* session.writeViaMux('/init\r'); // Send /init command
|
||||
* ```
|
||||
*/
|
||||
async writeViaMux(data: string): Promise<boolean> {
|
||||
this._trackSubmit(data);
|
||||
async writeViaMux(data: string, options: SessionWriteOptions = {}): Promise<boolean> {
|
||||
const submittedPrompt = this._trackSubmit(data, options);
|
||||
if (this._mux && this._muxSession) {
|
||||
return this._mux.sendInput(this.id, data);
|
||||
const sent = await this._mux.sendInput(this.id, data);
|
||||
if (sent) {
|
||||
this._emitSubmittedPrompt(submittedPrompt);
|
||||
this._verifySubmitted(data);
|
||||
}
|
||||
return sent;
|
||||
}
|
||||
// Fallback to PTY write
|
||||
if (this.ptyProcess) {
|
||||
this.ptyProcess.write(data);
|
||||
this._emitSubmittedPrompt(submittedPrompt);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Arm the composer check for a write that carried Enter (session-submit-verifier.ts):
|
||||
* Claude Code 2.1.277+ ignores Enter for the first 30-50 s after the composer paints,
|
||||
* so the pair `sendInput` just sent can leave the text stranded. Only a mux session
|
||||
* can read its pane, only text can be stranded, and the glyph is the CLI's own.
|
||||
*/
|
||||
private _verifySubmitted(data: string): void {
|
||||
if (!data.includes('\r') || !this._mux?.capturePaneText || !this._muxSession) return;
|
||||
const text = data.replace(/[\r\n]/g, '').trimEnd();
|
||||
if (!text) return;
|
||||
this._submitVerifier ??= new SubmitVerifier({
|
||||
capture: () =>
|
||||
this._isStopped || !this._mux || !this._muxSession
|
||||
? null
|
||||
: this._mux.capturePaneText?.(this._muxSession.muxName),
|
||||
sendEnter: () => this._mux?.sendInput(this.id, '\r'),
|
||||
// ⚠ NO fallback glyph here, unlike the screen-reading probe elsewhere in this file.
|
||||
// Only claude and codex declare a promptGlyph; the other eight modes would fall back
|
||||
// to claude's `❯`, which is ALSO starship's default shell prompt (and pure's, and
|
||||
// spaceship's, and p10k lean's). On a shell session the line `❯ npm run build` sits
|
||||
// on screen for as long as the command runs, promptStillInComposer() reads that as
|
||||
// "still unsubmitted", and the verifier presses Enter into the running program's
|
||||
// stdin on its 2s..60s schedule. Mostly a stray blank line; not harmless against a
|
||||
// y/N prompt, `read -p`, an installer or a pager, where it takes the default.
|
||||
// promptStillInComposer() returns undefined for an empty glyph, so this makes the
|
||||
// verifier inert for every CLI that does not declare one, which is what the Claude
|
||||
// Code 2.1.277 defect it exists for actually calls for.
|
||||
glyph: () => getCli(this.mode)?.capabilities.workDetect?.promptGlyph ?? '',
|
||||
log: (m) => console.log(`[Session ${this.id.slice(0, 8)}] ${m}`),
|
||||
});
|
||||
this._submitVerifier.arm(text);
|
||||
}
|
||||
|
||||
/** Current PTY dimensions — used to skip no-op resizes that trigger Ink redraws */
|
||||
private _ptyCols = 120;
|
||||
private _ptyRows = 40;
|
||||
|
||||
@@ -3482,6 +3482,14 @@ export class TmuxManager extends EventEmitter implements TerminalMultiplexer {
|
||||
{ encoding: 'utf-8', timeout: EXEC_TIMEOUT_MS }
|
||||
)
|
||||
);
|
||||
// Report the size the pane was really drawing at. The visible-frame path
|
||||
// below addresses every row absolutely, so a consumer whose terminal is
|
||||
// shorter than this piles the overflow rows onto its last line and loses
|
||||
// the rows it overwrote. The full-history path instead ends in a RELATIVE
|
||||
// cursor move, which costs it nothing when the two sizes disagree, so the
|
||||
// geometry is reported there for diagnosis rather than for repair. Only
|
||||
// the caller can see both sizes, so hand it this one.
|
||||
if (opts && geometry) opts.capturedGeometry = { cols: geometry.cols, rows: geometry.rows };
|
||||
|
||||
if (fullHistory) {
|
||||
// Without geometry there is no cursor move, so fall back to the old trim.
|
||||
|
||||
@@ -58,6 +58,24 @@ export type SessionMode =
|
||||
| 'deepseek'
|
||||
| 'omp';
|
||||
|
||||
/**
|
||||
* Who owns a session's name. `placeholder`: Codeman's own `w<n>-<case>` (or no
|
||||
* name at all), still eligible for auto-naming. `auto`: titled after its first
|
||||
* prompt (`w<n>-<case>: <title>`), which happens once. `manual`: set by a
|
||||
* person; auto-naming never touches it.
|
||||
*/
|
||||
export type SessionNameSource = 'placeholder' | 'auto' | 'manual';
|
||||
|
||||
/** Options for `Session.write()` / `Session.writeViaMux()`. */
|
||||
export interface SessionWriteOptions {
|
||||
/**
|
||||
* The bytes were typed by a person, or sent by an agent on their behalf
|
||||
* (browser keystrokes, `POST /api/sessions/:id/input`). Only such input can
|
||||
* name a tab; Ralph, respawn, cron and approval writes leave this unset.
|
||||
*/
|
||||
fromUser?: boolean;
|
||||
}
|
||||
|
||||
export type RemoteCommandMode = Extract<
|
||||
SessionMode,
|
||||
'shell' | 'claude' | 'opencode' | 'codex' | 'gemini' | 'antigravity' | 'pi' | 'grok' | 'deepseek' | 'omp'
|
||||
@@ -97,6 +115,25 @@ export interface RemoteHost extends RemoteSshOptions {
|
||||
username: string;
|
||||
port?: number;
|
||||
commands?: Partial<Record<RemoteCommandMode, string>>;
|
||||
/**
|
||||
* Optional Wake-on-LAN MAC address(es), comma-separated (e.g.
|
||||
* `04:d9:f5:80:c6:58`). Codeman sends the magic packet itself (UDP port 9
|
||||
* broadcast), so the common case needs no external script. A SLEEPING host's
|
||||
* port-22 probe still fails, which is what triggers the wake — this only
|
||||
* controls HOW the host is woken.
|
||||
*/
|
||||
wakeMac?: string;
|
||||
/**
|
||||
* Optional Wake-on-LAN command that powers this host on from SLEEP (e.g. a
|
||||
* wrapper script like `/home/joe/bin/whuff`). TAKES PRECEDENCE over `wakeMac`
|
||||
* (an explicit override for hosts that need a router/other-host wake). Absent
|
||||
* = no wake support and today's behavior exactly. Executed WITHOUT a shell (a
|
||||
* single executable path, never a command line), only from user input or an
|
||||
* explicit wake request on a session whose host is unreachable — never from
|
||||
* the auto-reconnect/boot-recovery path, which would re-wake a host seconds
|
||||
* after each suspend.
|
||||
*/
|
||||
wakeCommand?: string;
|
||||
}
|
||||
|
||||
export interface RemoteCase {
|
||||
@@ -137,6 +174,13 @@ export interface SessionRemote extends RemoteSshOptions {
|
||||
* session was created elsewhere. Only meaningful when `owned === false`.
|
||||
*/
|
||||
remoteSessionName?: string;
|
||||
/**
|
||||
* Wake-on-LAN command carried over from the host config (see `RemoteHost.wakeCommand`)
|
||||
* so the input route can wake a sleeping host without re-reading the host list.
|
||||
*/
|
||||
wakeCommand?: string;
|
||||
/** Wake-on-LAN MAC address(es) from the host config (see `RemoteHost.wakeMac`). */
|
||||
wakeMac?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -614,6 +658,8 @@ export interface SessionState {
|
||||
lastActivityAt: number;
|
||||
/** Session display name */
|
||||
name?: string;
|
||||
/** Who owns the name (see `SessionNameSource`); absent on states persisted before auto-naming existed. */
|
||||
nameSource?: SessionNameSource;
|
||||
/** Session mode */
|
||||
mode?: SessionMode;
|
||||
/** Auto-clear enabled */
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
*/
|
||||
|
||||
import type { Session } from '../../session.js';
|
||||
import type { SessionState } from '../../types.js';
|
||||
|
||||
export interface SessionPort {
|
||||
readonly sessions: ReadonlyMap<string, Session>;
|
||||
@@ -12,5 +13,40 @@ export interface SessionPort {
|
||||
setupSessionListeners(session: Session): Promise<void>;
|
||||
persistSessionState(session: Session): void;
|
||||
persistSessionStateNow(session: Session): void;
|
||||
/**
|
||||
* Re-apply the persisted state a freshly CONSTRUCTED session does not carry.
|
||||
*
|
||||
* A `Session` built from a record holds only what its constructor takes, so
|
||||
* persisting it would otherwise REPLACE the fuller record with the reduced one.
|
||||
* Two phases: `before-spawn` shapes the pane (the custom-model environment and
|
||||
* the nice priority) and must precede `startInteractive()`; `after-spawn` is
|
||||
* the session's own history (the pin, token and cost totals, auto-compact,
|
||||
* auto-clear, auto-resume, colour, image watcher, flicker filter) and must NOT
|
||||
* land on a session whose pane failed to start.
|
||||
*/
|
||||
reapplyPersistedSessionState(
|
||||
session: Session,
|
||||
saved: SessionState,
|
||||
phase: 'before-spawn' | 'after-spawn',
|
||||
options?: {
|
||||
/**
|
||||
* Re-arm a PENDING auto-resume schedule from the record's `autoResumeAt`.
|
||||
* Default true, which is what a Codeman restart wants: the limit footer
|
||||
* will not reprint on its own, so dropping the stamp there strands the
|
||||
* pause. A reboot restore passes false: the stamp predates the reboot,
|
||||
* the pane is new, and re-arming means every restored session types
|
||||
* `continue` into itself about a minute after one click. Auto-resume
|
||||
* stays ENABLED either way, so it re-arms on fresh evidence.
|
||||
*/
|
||||
rearmAutoResumeSchedule?: boolean;
|
||||
}
|
||||
): Promise<void>;
|
||||
/**
|
||||
* Undo a session that was registered but never got a working pane: the map
|
||||
* entry, its tab-layout slot, and any pane the launch created before throwing.
|
||||
* Unlike {@link cleanupSession} it leaves the persisted record, the lifetime
|
||||
* token totals, the Ralph state and the workspace's own files untouched.
|
||||
*/
|
||||
discardPartiallyBuiltSession(sessionId: string): Promise<void>;
|
||||
getSessionStateWithRespawn(session: Session): unknown;
|
||||
}
|
||||
|
||||
+331
-29
@@ -219,7 +219,8 @@ const _SSE_HANDLER_MAP = [
|
||||
// Remote auto-reconnect (COD-108)
|
||||
[SSE_EVENTS.REMOTE_SESSION_RECONNECTED, '_onRemoteSessionReconnected'],
|
||||
[SSE_EVENTS.REMOTE_RECONNECT_EXHAUSTED, '_onRemoteReconnectExhausted'],
|
||||
|
||||
[SSE_EVENTS.REMOTE_HOST_WAKING, '_onRemoteHostWaking'],
|
||||
[SSE_EVENTS.REMOTE_HOST_WAKE_FAILED, '_onRemoteHostWakeFailed'],
|
||||
// Ralph
|
||||
[SSE_EVENTS.SESSION_RALPH_LOOP_UPDATE, '_onRalphLoopUpdate'],
|
||||
[SSE_EVENTS.SESSION_RALPH_TODO_UPDATE, '_onRalphTodoUpdate'],
|
||||
@@ -549,6 +550,12 @@ class CodemanApp {
|
||||
// repaint-mode CLI pane, where tmux keeps no history of its own). The pull is
|
||||
// refused for those and retried far more slowly — see _maybeRefetchFullHistory.
|
||||
this._fullHistoryRepullUseless = new Set();
|
||||
// Sessions where the geometry replay has already been tried and did NOT
|
||||
// converge, so the pane is one this browser cannot size. Mirrors the Set
|
||||
// above: `resizeRetry` caps the recursion inside one select, and this is
|
||||
// what stops a fresh select from paying for the same answer again — see
|
||||
// the geometry gate in selectSession.
|
||||
this._geometryRetryUseless = new Set();
|
||||
this.terminalLoadStates = new Map(); // Map<sessionId, { generation, phase }>
|
||||
this.respawnStatus = {};
|
||||
this.respawnTimers = {}; // Track timed respawn timers
|
||||
@@ -957,6 +964,10 @@ class CodemanApp {
|
||||
this.registerServiceWorker();
|
||||
// Fetch tunnel status for header indicator (desktop only)
|
||||
this.loadTunnelStatus();
|
||||
// Ask whether a host reboot left sessions worth rebuilding (banner, never
|
||||
// automatic). handleInit() re-reads it on every SSE init; this covers the
|
||||
// path where that event never arrives.
|
||||
this.initRebootRestoreBanner?.();
|
||||
// Share a single settings fetch between both consumers
|
||||
const settingsPromise = fetch('/api/settings').then(r => r.ok ? r.json() : null).then(env => env?.data ?? null).catch(() => null);
|
||||
this.loadQuickStartCases(null, settingsPromise);
|
||||
@@ -1709,6 +1720,25 @@ class CodemanApp {
|
||||
console.error('[SSE] docker container recreated:', err);
|
||||
}
|
||||
});
|
||||
// Custom Model Endpoint Profiles: a session's own model got evicted on llama-swap by
|
||||
// another session's activity, detected AFTER the fact by a periodic server sweep (there
|
||||
// is no push notification from llama-swap itself) — see detectCustomModelSwapDisplacements
|
||||
// in custom-model-routes.ts. Global toast rather than a per-tab indicator: the displaced
|
||||
// session need not be the one currently open, and the whole point is telling the user
|
||||
// BEFORE they type into it expecting the model they picked.
|
||||
addListener(SSE_EVENTS.CUSTOM_MODEL_SWAPPED_OUT, (e) => {
|
||||
try {
|
||||
const d = e.data ? JSON.parse(e.data) : {};
|
||||
this.showToast(
|
||||
`${d.sessionName || d.sessionId}'s model (${d.previousModel}) was swapped out on llama-swap by another ` +
|
||||
`session — currently loaded: ${d.currentlyLoadedModel}. Sending a message there will reload it.`,
|
||||
'warning',
|
||||
{ duration: 0 }
|
||||
);
|
||||
} catch (err) {
|
||||
console.error('[SSE] custom model swapped out:', err);
|
||||
}
|
||||
});
|
||||
// Multi-user admin: live-refresh whichever admin views (panel/Users tab) are open.
|
||||
addListener(SSE_EVENTS.ADMIN_USERS_CHANGED, () => {
|
||||
window.codemanAdmin?.onUsersChanged?.();
|
||||
@@ -1814,6 +1844,9 @@ class CodemanApp {
|
||||
_onInit(data) {
|
||||
_crashDiag.log(`INIT: ${data.sessions?.length || 0} sessions`);
|
||||
this.handleInit(data);
|
||||
// Start the remote-host reachability poller even if no session switch follows
|
||||
// (a page loaded with the remote tab already active) — see host-wake-ui.js.
|
||||
this._ensureHostWakePoller?.();
|
||||
}
|
||||
|
||||
_onSessionCreated(data) {
|
||||
@@ -1906,6 +1939,53 @@ class CodemanApp {
|
||||
this._onSessionClearTerminal(data);
|
||||
}
|
||||
|
||||
/**
|
||||
* How a buffer load that just fetched `payload` must end.
|
||||
*
|
||||
* A tmux pane capture is a point-in-time frame, so nothing that reached the
|
||||
* browser after the response headers can already be in it. Such a load
|
||||
* replays exactly that tail; discarding it drops the CLI's output for the
|
||||
* rest of the load window, and its next partial redraw then lands on a frame
|
||||
* the terminal never received.
|
||||
*
|
||||
* A `history` payload is the server's byte buffer alone: the direct-PTY
|
||||
* fallback, or a mux pane whose capture came back empty. The route reads
|
||||
* that buffer in the same synchronous tick it takes the capture, so it is
|
||||
* current up to the route's own read and no further, which is the same
|
||||
* exposure. It deliberately keeps the pre-existing discard all the same:
|
||||
* both cases are rare, neither has been measured, and a duplicated Ink
|
||||
* redraw is more visible than a few milliseconds of missing output.
|
||||
* `capturedFromMux` below is the one line to widen if either turns out to
|
||||
* matter.
|
||||
*
|
||||
* `headersReceivedAt` is the caller's own `performance.now()` reading from
|
||||
* the moment the response arrived, compared only against other client-side
|
||||
* readings, so there is no clock skew to worry about.
|
||||
*
|
||||
* What this cutoff does NOT cover, and there are two contributors. The
|
||||
* server appends output to the byte buffer and emits it in the same tick,
|
||||
* but BROADCASTS on a batch timer (8ms over WebSocket, 16 to 50ms over SSE),
|
||||
* and the terminal route runs synchronously from `capture-pane` to its
|
||||
* return, so a batch already pending when the capture ran leaves the server
|
||||
* after the reply, arrives after `headersReceivedAt`, and is replayed
|
||||
* although the capture holds it. Separately, `captureActivePaneBuffer` is
|
||||
* `execSync`, which blocks the event loop for the whole capture: anything
|
||||
* tmux had already painted into the pane that the server had not yet read
|
||||
* from the attach PTY is in the capture too, is broadcast only after the
|
||||
* reply, and replays the same way. The duplicate is one batch interval plus
|
||||
* one capture wide, against a recovery window that spans the whole chunked
|
||||
* write. Closing it belongs on the server: flush that session's pending
|
||||
* batch before taking the capture.
|
||||
*
|
||||
* @param {{source?: string}} payload - The parsed `data` of a terminal response.
|
||||
* @param {number} headersReceivedAt - When that response reached this client.
|
||||
* @returns {{flushQueued: boolean, since: number}} Options for `_finishBufferLoad`.
|
||||
*/
|
||||
_bufferLoadFinishOpts(payload, headersReceivedAt) {
|
||||
const capturedFromMux = payload?.source === 'mux-visible' || payload?.source === 'mux-full-history';
|
||||
return { flushQueued: capturedFromMux, since: headersReceivedAt };
|
||||
}
|
||||
|
||||
_onSessionTerminal(data) {
|
||||
if (data.id === this.activeSessionId) {
|
||||
if (data.data.length > 32768) _crashDiag.log(`TERMINAL: ${(data.data.length/1024).toFixed(0)}KB`);
|
||||
@@ -1915,7 +1995,7 @@ class CodemanApp {
|
||||
// jump over the cap. Dropped data is recovered from the canonical buffer.
|
||||
const queued = (this.pendingWrites?.reduce((s, w) => s + w.length, 0) || 0)
|
||||
+ (this.flickerFilterBuffer?.length || 0)
|
||||
+ (this._loadBufferQueue?.reduce((s, w) => s + w.length, 0) || 0)
|
||||
+ (this._loadBufferQueue?.reduce((s, w) => s + w.data.length, 0) || 0)
|
||||
+ (this._terminalWriteInFlightBytes || 0);
|
||||
if (queued + data.data.length > 131072) { // 128KB — drop to prevent accumulation
|
||||
// Schedule a self-recovery once the
|
||||
@@ -2498,9 +2578,11 @@ class CodemanApp {
|
||||
? `/api/sessions/${sessionId}/terminal?full=1`
|
||||
: `/api/sessions/${sessionId}/terminal?tail=${TERMINAL_TAIL_SIZE}`
|
||||
);
|
||||
let headersReceivedAt = performance.now();
|
||||
let data = (await res.json())?.data ?? {};
|
||||
if (useFullHistory && data.terminalBuffer && this._replayWouldShrinkBuffer(data.terminalBuffer)) {
|
||||
res = await fetch(`/api/sessions/${sessionId}/terminal?tail=${TERMINAL_TAIL_SIZE}`);
|
||||
headersReceivedAt = performance.now();
|
||||
data = (await res.json())?.data ?? {};
|
||||
}
|
||||
// Bail on a tab switch mid-fetch: writing here would paint this session's
|
||||
@@ -2516,7 +2598,12 @@ class CodemanApp {
|
||||
const linesFromBottom = before ? Math.max(0, (before.baseY || 0) - (before.viewportY || 0)) : 0;
|
||||
this.terminal.clear();
|
||||
this.terminal.reset();
|
||||
await this.chunkedTerminalWrite(data.terminalBuffer);
|
||||
await this.chunkedTerminalWrite(
|
||||
data.terminalBuffer,
|
||||
TERMINAL_CHUNK_SIZE,
|
||||
undefined,
|
||||
this._bufferLoadFinishOpts(data, headersReceivedAt)
|
||||
);
|
||||
// A tail fetch can be partial, and the banner would otherwise keep
|
||||
// describing the pre-refresh buffer (#258).
|
||||
this._setHistoryTruncation(sessionId, data);
|
||||
@@ -2526,6 +2613,10 @@ class CodemanApp {
|
||||
});
|
||||
if (target === null || typeof this.terminal.scrollToLine !== 'function') this.terminal.scrollToBottom();
|
||||
else this.terminal.scrollToLine(target);
|
||||
// The load's own replay sampled the sticky-scroll baseline while the
|
||||
// terminal sat at the bottom of a just-rewritten buffer, so the next
|
||||
// flush would scroll back down and undo the restore above.
|
||||
this._syncStickyScrollBaseline();
|
||||
// Re-position local echo overlay at new prompt location
|
||||
this._localEchoOverlay?.rerender();
|
||||
// Resize PTY to match actual browser dimensions (critical for OpenCode
|
||||
@@ -2552,6 +2643,7 @@ class CodemanApp {
|
||||
// Fetch buffer, clear terminal, write buffer, resize (no Ctrl+L needed)
|
||||
try {
|
||||
const res = await fetch(`/api/sessions/${data.id}/terminal`);
|
||||
const headersReceivedAt = performance.now();
|
||||
const termData = (await res.json())?.data ?? {};
|
||||
|
||||
this.terminal.clear();
|
||||
@@ -2561,7 +2653,12 @@ class CodemanApp {
|
||||
// (markers don't help here - this is a static buffer reload, not live Ink redraws)
|
||||
const cleanBuffer = termData.terminalBuffer.replace(DEC_SYNC_STRIP_RE, '');
|
||||
// Use chunked write to avoid UI freeze with large buffers (can be 1-2MB)
|
||||
await this.chunkedTerminalWrite(cleanBuffer);
|
||||
await this.chunkedTerminalWrite(
|
||||
cleanBuffer,
|
||||
TERMINAL_CHUNK_SIZE,
|
||||
undefined,
|
||||
this._bufferLoadFinishOpts(termData, headersReceivedAt)
|
||||
);
|
||||
}
|
||||
|
||||
// Fire-and-forget resize — don't block on it
|
||||
@@ -3759,6 +3856,12 @@ class CodemanApp {
|
||||
// a fresh load / reconnect (authoritative; wins over the localStorage restore).
|
||||
if (data.planUsage) this.updatePlanUsageChip(data.planUsage);
|
||||
|
||||
// A board left open across a host reboot reconnects HERE, to a server that came
|
||||
// back with an empty session list. The reboot-restore offer is built at boot,
|
||||
// before any client could be listening, so re-read it on every init rather than
|
||||
// only on the page-load path.
|
||||
this.refreshRebootRestoreBanner?.();
|
||||
|
||||
// Update version displays (header and toolbar)
|
||||
if (data.version) {
|
||||
const versionEl = this.$('versionDisplay');
|
||||
@@ -5709,28 +5812,7 @@ class CodemanApp {
|
||||
if (ta) ta.dispatchEvent(new CompositionEvent('compositionend', { data: '' }));
|
||||
}
|
||||
} catch {}
|
||||
// Flush local echo text to PTY before switching tabs.
|
||||
// Send as a single batch (no Enter) so it lands in the session's readline
|
||||
// input buffer — avoids "old text resent on Enter" and overlay render bugs.
|
||||
// Track flushed length so _render() offsets the overlay correctly even before
|
||||
// the PTY echo arrives in the terminal buffer.
|
||||
if (this.activeSessionId) {
|
||||
const echoText = this._localEchoOverlay?.pendingText || '';
|
||||
// Include buffer-detected flushed text (from Tab completion, etc.)
|
||||
// so it's preserved across tab switches.
|
||||
const existingFlushed = this._localEchoOverlay?.getFlushed()?.count || 0;
|
||||
const existingFlushedText = this._localEchoOverlay?.getFlushed()?.text || '';
|
||||
if (echoText) {
|
||||
this._sendInputAsync(this.activeSessionId, echoText);
|
||||
}
|
||||
const totalOffset = existingFlushed + echoText.length;
|
||||
if (totalOffset > 0) {
|
||||
if (!this._flushedOffsets) this._flushedOffsets = new Map();
|
||||
if (!this._flushedTexts) this._flushedTexts = new Map();
|
||||
this._flushedOffsets.set(this.activeSessionId, totalOffset);
|
||||
this._flushedTexts.set(this.activeSessionId, existingFlushedText + echoText);
|
||||
}
|
||||
}
|
||||
this._flushLocalEchoTo(this.activeSessionId);
|
||||
this._localEchoOverlay?.clear();
|
||||
// Predictions are ephemeral + already sent: nothing to save/restore
|
||||
// across a tab switch (unlike the buffer overlay's setFlushed machinery)
|
||||
@@ -5745,6 +5827,45 @@ class CodemanApp {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Hand the local-echo overlay's unsent text to `sessionId` before anything
|
||||
* clears it, and record what has now been flushed so `_render()` offsets the
|
||||
* overlay correctly even before the PTY echo comes back.
|
||||
*
|
||||
* On a touch device the characters the user has typed live ONLY here until
|
||||
* Enter — they have never reached the PTY — so whoever clears the overlay
|
||||
* owes them a flush first. It is sent as one batch with no Enter, so it lands
|
||||
* in the session's readline buffer rather than submitting a line the user has
|
||||
* not finished.
|
||||
*
|
||||
* ⚠️ The session is a PARAMETER because the two callers are looking at
|
||||
* different ones. `_cleanupPreviousSession` flushes to the tab being left,
|
||||
* which is still `activeSessionId` when it runs. The `forceReload` branch in
|
||||
* `selectSession` flushes to the tab being RELOADED, and must do it before it
|
||||
* nulls `activeSessionId`: reading the field after that null is what silently
|
||||
* dropped the text, since the guard here then saw no session and the
|
||||
* unconditional `clear()` that follows took the characters with it.
|
||||
* @param {string|null} sessionId
|
||||
*/
|
||||
_flushLocalEchoTo(sessionId) {
|
||||
if (!sessionId) return;
|
||||
const echoText = this._localEchoOverlay?.pendingText || '';
|
||||
// Include buffer-detected flushed text (from Tab completion, etc.)
|
||||
// so it's preserved across tab switches.
|
||||
const existingFlushed = this._localEchoOverlay?.getFlushed()?.count || 0;
|
||||
const existingFlushedText = this._localEchoOverlay?.getFlushed()?.text || '';
|
||||
if (echoText) {
|
||||
this._sendInputAsync(sessionId, echoText);
|
||||
}
|
||||
const totalOffset = existingFlushed + echoText.length;
|
||||
if (totalOffset > 0) {
|
||||
if (!this._flushedOffsets) this._flushedOffsets = new Map();
|
||||
if (!this._flushedTexts) this._flushedTexts = new Map();
|
||||
this._flushedOffsets.set(sessionId, totalOffset);
|
||||
this._flushedTexts.set(sessionId, existingFlushedText + echoText);
|
||||
}
|
||||
}
|
||||
|
||||
_resetTerminalForReplay() {
|
||||
this.terminal.reset();
|
||||
this.terminal.write('\x1b[3J\x1b[H\x1b[2J');
|
||||
@@ -5852,7 +5973,12 @@ class CodemanApp {
|
||||
parsedAt,
|
||||
bufferLength: parsedBufferLength,
|
||||
completed,
|
||||
} = await this.chunkedTerminalWrite(buffer, TERMINAL_CHUNK_SIZE, sessionId);
|
||||
} = await this.chunkedTerminalWrite(
|
||||
buffer,
|
||||
TERMINAL_CHUNK_SIZE,
|
||||
sessionId,
|
||||
this._bufferLoadFinishOpts(payload, headersReceivedAt)
|
||||
);
|
||||
timing.resetAndParseMs = parsedAt - replayStartedAt;
|
||||
if (!completed || this.activeSessionId !== sessionId) return;
|
||||
// Keep shell tab restores bounded too. A user-triggered full-history pull
|
||||
@@ -5870,6 +5996,12 @@ class CodemanApp {
|
||||
const delta = parsedBufferLength - rowsBefore;
|
||||
if (delta > 0) this.terminal.scrollToLine(delta);
|
||||
else this.terminal.scrollToTop();
|
||||
// The load's own replay sampled the sticky-scroll baseline while the
|
||||
// terminal sat at the bottom of a just-rewritten buffer, so the next
|
||||
// flush would scroll back down and undo the restore above. This path is
|
||||
// reached only from a scroll-up gesture, so being dragged down is the
|
||||
// exact opposite of what the user asked for.
|
||||
this._syncStickyScrollBaseline();
|
||||
timing.totalMs = performance.now() - requestStartedAt;
|
||||
this._recordTerminalLoadTiming(timing);
|
||||
} catch {
|
||||
@@ -6008,6 +6140,13 @@ class CodemanApp {
|
||||
this._loadBufferQueue = null;
|
||||
this._terminalRefreshOwner = null;
|
||||
this._chunkedWriteGen = (this._chunkedWriteGen || 0) + 1;
|
||||
// Anything typed but not yet submitted lives in the local-echo overlay and
|
||||
// has never reached the PTY. `_cleanupPreviousSession` below flushes it,
|
||||
// but only for a session it can still see, and the null on the next line
|
||||
// hides this one from it. Flush first or the characters are cleared
|
||||
// unread. The geometry replay re-enters here with no gesture behind it,
|
||||
// so on a touch device this fires while the user is still typing.
|
||||
this._flushLocalEchoTo(sessionId);
|
||||
this.activeSessionId = null;
|
||||
}
|
||||
// Focus terminal SYNCHRONOUSLY before any await — iOS Safari only honors
|
||||
@@ -6074,6 +6213,9 @@ class CodemanApp {
|
||||
// bar (issue #262). Also disarms a one-shot Ctrl left over from the tab we
|
||||
// just left, so it can never fire against the session we just opened.
|
||||
if (typeof KeyboardAccessoryBar !== 'undefined') KeyboardAccessoryBar.refreshForActiveSession();
|
||||
// Remote-host reachability banner: only meaningful for a remote session, so this
|
||||
// also clears it when the newly active tab is local.
|
||||
this.refreshHostWakeBanner?.(sessionId);
|
||||
|
||||
// Restore flushed offset AND text IMMEDIATELY so backspace/typing work during
|
||||
// the async buffer load. Without this, the offset is 0 during the
|
||||
@@ -6187,6 +6329,10 @@ class CodemanApp {
|
||||
// sendResize is a no-op on the server when dims haven't changed, so
|
||||
// calling it every tab switch is cheap.
|
||||
const dimsChanged = await this.sendResize(sessionId, { forceHttp: true }).catch(() => false);
|
||||
// The size the capture below will be taken against. The debounced resize
|
||||
// handler can move the terminal again while the load runs, so this is a
|
||||
// recorded value rather than a later read of `_lastResizeDims`.
|
||||
const dimsAtCapture = this.getTerminalDimensions?.();
|
||||
if (this._isStaleSelect(selectGen)) {
|
||||
this._clearTerminalLoadState(sessionId, selectGen);
|
||||
return;
|
||||
@@ -6311,6 +6457,15 @@ class CodemanApp {
|
||||
}
|
||||
const data = (await res.json())?.data ?? {};
|
||||
const bodyParsedAt = performance.now();
|
||||
// How this load must end, decided here because `chunkedTerminalWrite` is
|
||||
// what actually ends it for a non-empty buffer. A tmux pane capture is a
|
||||
// point-in-time frame, so nothing that reached the browser after the
|
||||
// response headers can already be in it. Replay exactly that tail;
|
||||
// discarding it drops the CLI's output for the rest of the load window,
|
||||
// and its next partial redraw then lands on a frame the terminal never
|
||||
// received. `since` keeps the pre-capture events dropped, because the
|
||||
// capture does hold those and replaying them would duplicate output.
|
||||
const finishOpts = this._bufferLoadFinishOpts(data, headersReceivedAt);
|
||||
_crashDiag.log(`FETCH_DONE: ${data.terminalBuffer ? (data.terminalBuffer.length/1024).toFixed(0) + 'KB' : 'empty'} truncated=${data.truncated}`);
|
||||
|
||||
let freshResetAndParseMs = 0;
|
||||
@@ -6337,7 +6492,8 @@ class CodemanApp {
|
||||
const { parsedAt: freshParsedAt } = await this.chunkedTerminalWrite(
|
||||
data.terminalBuffer,
|
||||
TERMINAL_CHUNK_SIZE,
|
||||
bufferLoadOwner
|
||||
bufferLoadOwner,
|
||||
finishOpts
|
||||
);
|
||||
freshResetAndParseMs = freshParsedAt - replayStartedAt;
|
||||
if (this._isStaleSelect(selectGen)) {
|
||||
@@ -6387,7 +6543,14 @@ class CodemanApp {
|
||||
// COD-144: when the load painted nothing, FLUSH the queued events instead of
|
||||
// discarding — a new session's prompt arrives only as a queued SSE event.
|
||||
if (this._isLoadingBuffer) {
|
||||
this._finishBufferLoad(bufferLoadOwner, { flushQueued: bufferWasEmpty });
|
||||
// Only reached when the write was skipped. COD-144 lives here: a new
|
||||
// session's first prompt exists only as a queued event that predates the
|
||||
// response, so an empty paint replays its queue WHOLE rather than from
|
||||
// the header timestamp.
|
||||
this._finishBufferLoad(
|
||||
bufferLoadOwner,
|
||||
bufferWasEmpty ? { flushQueued: true, since: 0 } : finishOpts
|
||||
);
|
||||
}
|
||||
// Drop the guard so user input clears state normally
|
||||
this._restoringFlushedState = false;
|
||||
@@ -6418,6 +6581,84 @@ class CodemanApp {
|
||||
// annoyance that disappear on the user's next keypress; data loss is not
|
||||
// acceptable. Do NOT re-introduce Ctrl+L here.
|
||||
this.sendResize(sessionId);
|
||||
// sendResize fits synchronously before its first await, so this reads the
|
||||
// size that survived the load rather than the one the capture was taken
|
||||
// at. The two differ whenever the terminal was still settling.
|
||||
const dimsAfterLoad = this.getTerminalDimensions?.();
|
||||
// Only a visible-frame capture positions its rows absolutely, and only
|
||||
// that frame can be damaged by a terminal of the wrong size. A `full=1`
|
||||
// body is linear scrollback closed by a RELATIVE cursor move
|
||||
// (`formatCursorRestore`), which is relative precisely so the browser's
|
||||
// row count need not match the pane's, and a `history` body is the byte
|
||||
// stream, which carries no row alignment to protect. Replaying either at
|
||||
// a different size repairs nothing, and the full-history replay costs a
|
||||
// second whole-scrollback capture to learn that. Since the first select
|
||||
// of every non-shell session per page takes the full-history path, an
|
||||
// ungated comparison fires most often on the one response it cannot help.
|
||||
const framePositionsRowsAbsolutely = data.source === 'mux-visible';
|
||||
// `mux-visible` is necessary but not sufficient: when the `display-message`
|
||||
// cursor query fails, `capturePaneBuffer` skips the snapshot repaint and
|
||||
// returns the raw capture, and the route still labels a non-empty body
|
||||
// `mux-visible`. That body positions nothing and reports no geometry, so a
|
||||
// size that moved during such a load has nothing to repair, and replaying
|
||||
// would buy a second capture, a reset plus chunked rewrite, a dropped
|
||||
// WebSocket and a discarded xterm snapshot for it. The two comparisons
|
||||
// below already stand down on an absent field; this one has to as well.
|
||||
const sizeMovedUnderLoad =
|
||||
framePositionsRowsAbsolutely &&
|
||||
Number.isFinite(data.captureRows) &&
|
||||
!!dimsAtCapture &&
|
||||
!!dimsAfterLoad &&
|
||||
(dimsAfterLoad.cols !== dimsAtCapture.cols || dimsAfterLoad.rows !== dimsAtCapture.rows);
|
||||
// A capture positions every row absolutely, so a pane taller than this
|
||||
// terminal writes its overflow rows onto the last line and loses the rows
|
||||
// it overwrote. A pane WIDER than this terminal damages the same frame a
|
||||
// second way: `formatPaneSnapshot` paints each row out to the pane's own
|
||||
// width, so a narrower browser wraps every painted row, and the wrap on
|
||||
// the last one scrolls the whole frame up by a row. Both happen when the
|
||||
// capture wins a race against the resize meant to precede it, which is
|
||||
// what the retry below repairs.
|
||||
//
|
||||
// It also happens when `Session.resize` DECLINED the resize, which it does
|
||||
// for a small viewport while a desktop viewport's size claim is live. The
|
||||
// retry cannot repair that one: it re-sends the same declined resize and
|
||||
// captures the same too-tall pane. `resizeRetry` stops it after the one
|
||||
// extra attempt, and the frame is shown as-is. Repairing that case means
|
||||
// changing who owns the pane size, which is a policy question this does
|
||||
// not touch. What the flag does buy there is that the client can SEE the
|
||||
// mismatch at all, which it previously could not.
|
||||
//
|
||||
// An ABSENT field is not a fit. It means the capture reported no geometry
|
||||
// at all, so nothing was positioned and there is nothing to repair.
|
||||
const capturedTallerThanTerminal =
|
||||
framePositionsRowsAbsolutely &&
|
||||
Number.isFinite(data.captureRows) &&
|
||||
data.captureRows > (this.terminal?.rows || 0);
|
||||
const capturedWiderThanTerminal =
|
||||
framePositionsRowsAbsolutely &&
|
||||
Number.isFinite(data.captureCols) &&
|
||||
data.captureCols > (this.terminal?.cols || 0);
|
||||
// The retry replays at `dimsAfterLoad`, so it can only change what is on
|
||||
// screen if the pane was drawing at some OTHER size. When the reported
|
||||
// geometry already IS that size, the second pass captures the identical
|
||||
// frame and pays a full reload to do it: another fetch, another
|
||||
// `_resetTerminalForReplay()` and chunked rewrite (a visible re-flash),
|
||||
// and, because it goes through `forceReload`, a dropped and reopened
|
||||
// WebSocket plus a deleted xterm snapshot.
|
||||
//
|
||||
// That equality is the signature of a CLAMP rather than a race.
|
||||
// `getTerminalDimensions()` floors at 40x10 while `fitAddon.fit()` does
|
||||
// not, so a terminal narrower than 40 columns or shorter than 10 rows
|
||||
// reports a pane permanently bigger than itself, and every select would
|
||||
// retry without ever converging. A race never produces this equality: its
|
||||
// whole premise is that the pane was still at the size we asked it to
|
||||
// leave. The other non-converging case, `Session.resize` declining a
|
||||
// small viewport while a desktop claim is live, does not produce it
|
||||
// either — that pane sits at the DESKTOP's size — so it still costs the
|
||||
// one capped attempt, and stopping it needs the pane-ownership policy
|
||||
// this does not touch.
|
||||
const captureMatchesRequestedSize =
|
||||
!!dimsAfterLoad && data.captureCols === dimsAfterLoad.cols && data.captureRows === dimsAfterLoad.rows;
|
||||
|
||||
// Defer secondary panel updates so they don't block the main thread
|
||||
// after terminal content is already visible.
|
||||
@@ -6518,6 +6759,67 @@ class CodemanApp {
|
||||
this._clearTerminalLoadState(sessionId, selectGen);
|
||||
_crashDiag.log(`SELECT_DONE: ${selectDoneMs.toFixed(0)}ms`);
|
||||
console.log(`[CRASH-DIAG] selectSession DONE: ${sessionId.slice(0,8)} in ${selectDoneMs.toFixed(0)}ms`);
|
||||
// Remember whether the replay was worth it, because `resizeRetry` only
|
||||
// caps the recursion INSIDE one select and says nothing about the next
|
||||
// one. A pane this browser cannot size — one whose resize `Session.resize`
|
||||
// declines while a desktop claim is live, or one a second tmux client is
|
||||
// also holding — reports the same mismatch on every select, so without a
|
||||
// memo the diagnosis is paid for again on every tab switch, forever: two
|
||||
// fetches per select rather than one. Each extra pass costs a second
|
||||
// `capture-pane`, which is `execSync` and blocks the server's event loop,
|
||||
// plus a reset and chunked rewrite, a discarded snapshot and cache entry,
|
||||
// and a dropped and reopened WebSocket.
|
||||
//
|
||||
// A retry pass that STILL does not fit is the proof, since the retry ran
|
||||
// at the size that stuck and the pane ignored it. Geometry that fits
|
||||
// clears the memo, so a pane that becomes sizeable again (the desktop tab
|
||||
// closes, the claim goes idle) is repaired on the next select. The race
|
||||
// case is untouched: it converges on its first attempt, so it never
|
||||
// reaches the branch that latches.
|
||||
const capturedGeometryFits =
|
||||
framePositionsRowsAbsolutely &&
|
||||
Number.isFinite(data.captureRows) &&
|
||||
!capturedTallerThanTerminal &&
|
||||
!capturedWiderThanTerminal;
|
||||
if (capturedGeometryFits) {
|
||||
this._geometryRetryUseless?.delete(sessionId);
|
||||
} else if (options?.resizeRetry && (capturedTallerThanTerminal || capturedWiderThanTerminal)) {
|
||||
(this._geometryRetryUseless ||= new Set()).add(sessionId);
|
||||
}
|
||||
// What is on screen was drawn for a geometry this terminal does not have.
|
||||
// Replaying once against the size that stuck is the only thing that
|
||||
// repairs it: SIGWINCH reaches the CLI only on a real size change, and
|
||||
// the pane is already at its final size, so no redraw is coming.
|
||||
// `resizeRetry` caps this at one attempt, so two competing fits cannot
|
||||
// trade replays forever.
|
||||
if (
|
||||
(sizeMovedUnderLoad || capturedTallerThanTerminal || capturedWiderThanTerminal) &&
|
||||
!captureMatchesRequestedSize &&
|
||||
!this._geometryRetryUseless?.has(sessionId) &&
|
||||
!options?.resizeRetry &&
|
||||
!this._isStaleSelect(selectGen)
|
||||
) {
|
||||
_crashDiag.log(
|
||||
`RESIZE_RETRY: capture ${data.captureCols}x${data.captureRows} vs terminal ` +
|
||||
`${this.terminal?.cols}x${this.terminal?.rows}` +
|
||||
(sizeMovedUnderLoad ? ' (size moved under load)' : '')
|
||||
);
|
||||
// Re-arm the full-history pull ONLY if this pass actually used one, so
|
||||
// the retry replays the same content at the geometry that stuck. A pass
|
||||
// that took the bounded tail must retry on the tail too: clearing the
|
||||
// flag unconditionally would UPGRADE a tab switch into a fresh
|
||||
// multi-megabyte scrollback capture it never asked for.
|
||||
//
|
||||
// UNREACHABLE as written, and kept for the invariant rather than the
|
||||
// branch. A `useFullHistory` pass sends `full=1`, and the route answers
|
||||
// `full=1` with `mux-full-history` or `history`, never `mux-visible`
|
||||
// (see the source ladder in session-routes.ts), so the gate above
|
||||
// already rules out every pass that consumed the flag. Do not read this
|
||||
// line as evidence that a page load retries: it does not, and the test
|
||||
// suite pins that it does not.
|
||||
if (useFullHistory) this._fullHistoryLoaded.delete(sessionId);
|
||||
await this.selectSession(sessionId, { auto: true, forceReload: true, resizeRetry: true });
|
||||
}
|
||||
} catch (err) {
|
||||
if (this._isLoadingBuffer) this._finishBufferLoad(bufferLoadOwner);
|
||||
this._restoringFlushedState = false;
|
||||
|
||||
@@ -806,6 +806,71 @@ function decideAutoCopy({ enabled, text, lastCopied, pending } = {}) {
|
||||
return 'copy';
|
||||
}
|
||||
|
||||
// The text a copy should put on the clipboard, given xterm's raw selection.
|
||||
// Pure: the caller reads the selection and decides the mode, this transforms.
|
||||
//
|
||||
// xterm hands back whole screen ROWS, and its own trim only drops cells that
|
||||
// were never written to. A full-screen TUI writes real spaces across the part
|
||||
// of a row it is not using, so that padding counts as content and rides along
|
||||
// to the clipboard: measured against Claude Code in a 282-column pane, single
|
||||
// lines arrived carrying 138 trailing spaces. Native terminals trim it on copy
|
||||
// (Windows Terminal, iTerm2 and GNOME Terminal all do), decideAutoCopy above
|
||||
// already calls a wall of spaces "never what the gesture meant", and
|
||||
// _selectTouchSelectionLine already treats those cells as padding. This is that
|
||||
// same rule for the mouse and keyboard paths, which never had it.
|
||||
//
|
||||
// ⚠ Trailing padding ONLY. A shared LEADING indent is deliberately left alone,
|
||||
// and this note is here so the idea is not re-derived: it was built, measured
|
||||
// and dropped before merge. Removing the longest leading run every selected row
|
||||
// shares looks like the mirror image of the trailing trim and is not, because
|
||||
// no native terminal does it and the transform cannot tell a TUI's margin from
|
||||
// content that is genuinely indented. Measured over 401 445 three-row windows
|
||||
// across 1 010 tracked files in this repo, it fired on 73% of them: 92% inside
|
||||
// a YAML workflow, 76% over `git log` output, 48% in a TypeScript source file.
|
||||
// No width threshold separates the two, because they are the same widths: a
|
||||
// live Claude Code pane's own margins measure 2 and 5 columns while the most
|
||||
// common non-TUI shared run is 4, sitting between them.
|
||||
//
|
||||
// The asymmetry that settles it is in the failure modes. A wrong trailing trim
|
||||
// costs nothing. A wrong dedent silently deletes information that was on the
|
||||
// screen, with no signal to the user and nothing in the clipboard to hint at
|
||||
// it, and it is wrong on `git log` bodies, on indented code read out of `cat`
|
||||
// (semantic in Python), on `git diff` context rows where the leading space is
|
||||
// the marker, and on stack traces.
|
||||
//
|
||||
// ⚠ It also cannot be made consistent cheaply. Whether the first row joins the
|
||||
// measurement depended on the mousedown COLUMN, which the user never sees, so
|
||||
// one block of three rows produced three different clipboard results; and the
|
||||
// flag read `getSelectionPosition().start`, which is the mousedown anchor that
|
||||
// xterm never normalises, so dragging UP through a block read it off the bottom
|
||||
// row. If it is ever revisited, the one qualification that measured clean is
|
||||
// painted trailing padding (a full-screen TUI writes real spaces across every
|
||||
// row; a shell pane leaves those cells never-written, so xterm trims them):
|
||||
// zero false positives over all 401 445 windows. It still mangles a `git log`
|
||||
// body sitting inside an agent's own gutter, which is why it was not taken now.
|
||||
function cleanCopiedSelection(text) {
|
||||
if (typeof text !== 'string' || !text) return '';
|
||||
// Split on \n and leave any \r in place: xterm joins rows with \r\n on
|
||||
// Windows, and the clipboard should keep the endings xterm chose.
|
||||
// Scanned rather than matched. A selection can run to the 50 000-row
|
||||
// scrollback ceiling, and `/[ \t]+(\r?)$/` is QUADRATIC on a line whose spaces
|
||||
// are followed by any non-space character, which is what right-aligned or
|
||||
// centred TUI content looks like: the engine retries the run from every
|
||||
// whitespace position and backtracks over it. Measured over 50 000 rows with a
|
||||
// 280-column run, that regex took 2.9s against 1.3ms for the scan below, and a
|
||||
// 2 000-column run took 16s. It is also the faster of the two on an ordinary
|
||||
// padded row. A length is returned rather than a trimmed string so a
|
||||
// \r-terminated line costs no substring either.
|
||||
const trimEnd = (line) => {
|
||||
let end = line.length;
|
||||
if (end > 0 && line[end - 1] === '\r') end--;
|
||||
let cut = end;
|
||||
while (cut > 0 && (line[cut - 1] === ' ' || line[cut - 1] === '\t')) cut--;
|
||||
return cut === end ? line : line.slice(0, cut) + line.slice(end);
|
||||
};
|
||||
return text.split('\n').map(trimEnd).join('\n');
|
||||
}
|
||||
|
||||
if (typeof window !== 'undefined') {
|
||||
window.WEBGL_FALLBACK = WEBGL_FALLBACK;
|
||||
window.evaluateWebGLLongTaskTrip = evaluateWebGLLongTaskTrip;
|
||||
@@ -854,6 +919,9 @@ if (typeof window !== 'undefined') {
|
||||
decide: decideAutoCopy,
|
||||
MAX_CHARS: AUTO_COPY_MAX_CHARS,
|
||||
};
|
||||
window.CodemanCopySelection = {
|
||||
clean: cleanCopiedSelection,
|
||||
};
|
||||
window.CodemanTerminalFont = {
|
||||
DEFAULT_STACK: TERMINAL_FONT_DEFAULT_STACK,
|
||||
resolve: resolveTerminalFontFamily,
|
||||
@@ -1057,6 +1125,9 @@ const SSE_EVENTS = {
|
||||
REMOTE_SESSION_DROPPED: 'remote:sessionDropped',
|
||||
REMOTE_SESSION_RECONNECTED: 'remote:sessionReconnected',
|
||||
REMOTE_RECONNECT_EXHAUSTED: 'remote:reconnectExhausted',
|
||||
// Wake-on-LAN from user input on a sleeping remote host
|
||||
REMOTE_HOST_WAKING: 'remote:hostWaking',
|
||||
REMOTE_HOST_WAKE_FAILED: 'remote:hostWakeFailed',
|
||||
|
||||
// Ralph
|
||||
SESSION_RALPH_LOOP_UPDATE: 'session:ralphLoopUpdate',
|
||||
@@ -1094,6 +1165,9 @@ const SSE_EVENTS = {
|
||||
APPROVAL_UPDATED: 'approval:updated',
|
||||
APPROVAL_RESOLVED: 'approval:resolved',
|
||||
|
||||
// Custom Model Endpoint Profiles
|
||||
CUSTOM_MODEL_SWAPPED_OUT: 'custom-model:swapped-out',
|
||||
|
||||
// Subagents (Claude Code background agents)
|
||||
SUBAGENT_DISCOVERED: 'subagent:discovered',
|
||||
SUBAGENT_UPDATED: 'subagent:updated',
|
||||
|
||||
@@ -0,0 +1,440 @@
|
||||
/**
|
||||
* @fileoverview Remote-host wake-on-LAN: the "host unreachable" banner + its config dialog.
|
||||
*
|
||||
* A sleeping remote host does not fail loudly. The local tmux pane runs `ssh`, and when
|
||||
* the machine suspends, that ssh child stalls: `tmux send-keys` still SUCCEEDS, so typed
|
||||
* input disappears with no error and the pane looks alive. The server side
|
||||
* (`src/remote-wake.ts`) buffers input and wakes the host when the user types; this
|
||||
* module makes the state VISIBLE and gives it a button, which is what turns "why is
|
||||
* nothing happening" into one click.
|
||||
*
|
||||
* Behavior:
|
||||
* - Asks `GET /api/sessions/:id/reachability` for the ACTIVE remote session only:
|
||||
* once when the tab is activated (a user action), and every `POLL_MS` while the tab
|
||||
* is visible ONLY for a host with a wake target. The timer is the one thing here that
|
||||
* is not user-driven, and each poll is a TCP connect to the host — the same
|
||||
* timer-driven traffic invariant #2 rejects keepalives for: it cannot wake a host,
|
||||
* but it can keep an activity-based suspend timer from firing. So a host Codeman
|
||||
* could not wake anyway is never polled on a timer. A host behind a jump host or
|
||||
* SOCKS proxy (`probeable: false`) is never polled at all: the probe cannot reach
|
||||
* it, so its answer would only ever be a false "asleep". The endpoint shares the
|
||||
* server's probe cache with the input path, so opening the tab also primes the
|
||||
* wake path.
|
||||
* - Unreachable + a configured wake target → "Wake" button → `POST /api/sessions/:id/wake`
|
||||
* (which wakes, waits, reattaches the pane and flushes buffered input).
|
||||
* - Unreachable + NO wake target → "Configure WoL" → `#wakeConfigModal`, a small form
|
||||
* for this host's MAC/command that saves via `PUT /api/remote-hosts/:id`. The server
|
||||
* re-resolves host config while the session is live, so saving takes effect without
|
||||
* restarting the session.
|
||||
* - SSE (`remote:hostWaking`, `remote:hostWakeFailed`, `remote:sessionReconnected`)
|
||||
* keeps the banner in sync while a wake is running.
|
||||
*
|
||||
* @mixin Extends CodemanApp.prototype via Object.assign
|
||||
* @dependency app.js (CodemanApp class, this.sessions, this.activeSessionId, showToast)
|
||||
* @dependency constants.js (SSE_EVENTS — the remote:hostWaking / remote:hostWakeFailed names)
|
||||
* @loadorder 12.2 — loaded after session-ui.js, before webview-tabs.js
|
||||
*/
|
||||
|
||||
const HOST_WAKE_POLL_MS = 30_000;
|
||||
|
||||
Object.assign(CodemanApp.prototype, {
|
||||
/** Per-tab banner state (single active session at a time). */
|
||||
_hostWake: null,
|
||||
/** The page-wide poller interval (created once, see `_ensureHostWakePoller`). */
|
||||
_hostWakeTimer: null,
|
||||
|
||||
/** Fresh state for a session we just switched to. */
|
||||
_hostWakeState() {
|
||||
return {
|
||||
sessionId: null,
|
||||
/** Last reachability answer, or null before the first poll. */
|
||||
reachable: null,
|
||||
/** 'command' | 'mac' | 'none' — what the banner action should do. */
|
||||
wakeConfigured: 'none',
|
||||
host: '',
|
||||
label: '',
|
||||
/**
|
||||
* False for a host the server's probe cannot reach (behind a jump host or SOCKS
|
||||
* proxy): its reachability is unknown, so there is no banner and no polling.
|
||||
*/
|
||||
probeable: true,
|
||||
/** True between clicking Wake and the answer coming back. */
|
||||
waking: false,
|
||||
/**
|
||||
* True only when the server is actually holding bytes for this session (the typing
|
||||
* path buffers them). Browser keystrokes go over the WebSocket, which never passes
|
||||
* through the wake registry — so the Wake BUTTON must not claim input is queued.
|
||||
*/
|
||||
queuedInput: false,
|
||||
/** Set when the last wake attempt or poll failed. */
|
||||
error: '',
|
||||
};
|
||||
},
|
||||
|
||||
/**
|
||||
* Entry point from the session switcher — called for every active session, remote or
|
||||
* not, so it must be cheap and must clear the banner for local sessions.
|
||||
*
|
||||
* ⚠️ The POLLER is page-wide and independent of this call on purpose: a session
|
||||
* switch is not the only way the active tab changes (boot restore, a page loaded with
|
||||
* the tab already active, and `selectSession`'s own early return for the tab you are
|
||||
* already on), and the banner must not depend on any single one of those paths
|
||||
* running — that is exactly how it could silently never appear.
|
||||
*/
|
||||
refreshHostWakeBanner(sessionId) {
|
||||
this._ensureHostWakePoller();
|
||||
const state = this._hostWake;
|
||||
if (state && state.sessionId && state.sessionId !== sessionId) this._hostWake = null;
|
||||
this._hostWakeTick();
|
||||
},
|
||||
|
||||
/** Create the page-wide poller once (interval + a visibility wake-up). */
|
||||
_ensureHostWakePoller() {
|
||||
if (this._hostWakeTimer) return;
|
||||
this._hostWakeTimer = setInterval(() => this._hostWakeTick({ periodic: true }), HOST_WAKE_POLL_MS);
|
||||
document.addEventListener('visibilitychange', () => {
|
||||
if (document.visibilityState === 'visible') this._hostWakeTick({ periodic: true });
|
||||
});
|
||||
},
|
||||
|
||||
/**
|
||||
* One poller tick: resolve the ACTIVE session, reset the banner when it changed, and
|
||||
* ask the server. No-op while the page is hidden (a background tab must not poll).
|
||||
*
|
||||
* `periodic` marks the timer (and the visibility wake-up) as opposed to a tab
|
||||
* activation: a periodic tick polls only a host with a wake target, see the module
|
||||
* comment. The activation poll is what still offers "Configure WoL" for a sleeping
|
||||
* host that has none — one connect, on a user action.
|
||||
*/
|
||||
_hostWakeTick({ periodic = false } = {}) {
|
||||
if (typeof document !== 'undefined' && document.visibilityState === 'hidden') return;
|
||||
const sessionId = this.activeSessionId;
|
||||
const session = sessionId && this.sessions ? this.sessions.get(sessionId) : null;
|
||||
if (!sessionId || !session || !session.remote) {
|
||||
// Render unconditionally: `refreshHostWakeBanner` clears `_hostWake` BEFORE
|
||||
// calling this tick, so a guard here would skip the repaint and leave the
|
||||
// banner up on every chat (the clear and the repaint must not be coupled to
|
||||
// whoever cleared the state). Idempotent — with a null state it just hides.
|
||||
this._hostWake = null;
|
||||
this._renderHostWakeBanner();
|
||||
return;
|
||||
}
|
||||
let state = this._hostWake;
|
||||
let fresh = false;
|
||||
if (!state || state.sessionId !== sessionId) {
|
||||
fresh = true;
|
||||
state = this._hostWake = this._hostWakeState();
|
||||
state.sessionId = sessionId;
|
||||
state.host = session.remote.host || '';
|
||||
state.label = session.remote.label || 'Remote host';
|
||||
// Text from the session payload first (instant, no round trip), corrected by the
|
||||
// poll — a session whose wake config was added after launch only knows it after
|
||||
// the server resolves host config. The kind matters: the payload can say WHICH
|
||||
// path is configured, so a command-only host is not mislabelled 'mac' until the
|
||||
// first poll lands.
|
||||
state.wakeConfigured = session.remote.wakeMac ? 'mac' : session.remote.wakeCommand ? 'command' : 'none';
|
||||
// Known from the payload already: a proxied host is not probeable (the server
|
||||
// says so too, on every answer), so not even the activation poll is worth a
|
||||
// round trip whose verdict could only be a wrong "asleep".
|
||||
state.probeable = !(session.remote.jumpHost || session.remote.socksProxy);
|
||||
this._renderHostWakeBanner();
|
||||
}
|
||||
if (!state.probeable) return;
|
||||
if (periodic && !fresh && state.wakeConfigured === 'none') return;
|
||||
this._pollHostReachability();
|
||||
},
|
||||
|
||||
/** One reachability check for the active remote session. */
|
||||
async _pollHostReachability(force = false) {
|
||||
const state = this._hostWake;
|
||||
if (!state || !state.sessionId) return;
|
||||
const sessionId = state.sessionId;
|
||||
try {
|
||||
const res = await fetch(`/api/sessions/${encodeURIComponent(sessionId)}/reachability${force ? '?force=1' : ''}`);
|
||||
const data = await res.json();
|
||||
if (!data.success) return;
|
||||
// The tab may have changed while this was in flight.
|
||||
if (this._hostWake !== state || state.sessionId !== sessionId) return;
|
||||
// `reachable` is `null` (unknown, not unreachable) for a host the probe cannot
|
||||
// reach — only a PROVEN `false` may raise the banner.
|
||||
state.reachable = data.data.reachable !== false;
|
||||
if (data.data.probeable === false) state.probeable = false;
|
||||
state.wakeConfigured = data.data.wakeConfigured || 'none';
|
||||
if (data.data.host) state.host = data.data.host;
|
||||
if (data.data.label) state.label = data.data.label;
|
||||
if (state.reachable) {
|
||||
state.waking = false;
|
||||
state.error = '';
|
||||
}
|
||||
this._renderHostWakeBanner();
|
||||
} catch {
|
||||
/* A failed poll is not a state change: leave the banner as it was. */
|
||||
}
|
||||
},
|
||||
|
||||
/** Draw the banner from `_hostWake`. */
|
||||
_renderHostWakeBanner() {
|
||||
const state = this._hostWake;
|
||||
const banner = this.$('hostWakeBanner');
|
||||
const text = this.$('hostWakeBannerText');
|
||||
const detail = this.$('hostWakeBannerDetail');
|
||||
const action = this.$('hostWakeBannerAction');
|
||||
if (!banner || !text || !action) return;
|
||||
|
||||
const visible = Boolean(state && state.sessionId && state.reachable === false);
|
||||
banner.hidden = !visible;
|
||||
if (!visible) return;
|
||||
|
||||
const hasTarget = state.wakeConfigured !== 'none';
|
||||
const target = state.label || state.host || 'Remote host';
|
||||
if (state.waking) {
|
||||
text.textContent = `Waking ${target} …`;
|
||||
} else if (state.error) {
|
||||
text.textContent = `${target} did not wake up`;
|
||||
} else {
|
||||
text.textContent = `${target} is not reachable`;
|
||||
}
|
||||
if (detail) {
|
||||
detail.textContent = state.waking
|
||||
? state.queuedInput
|
||||
? 'input is queued until it is back'
|
||||
: 'waiting for the host to come back'
|
||||
: hasTarget
|
||||
? `ssh ${state.host}`
|
||||
: 'no wake-on-LAN configured';
|
||||
}
|
||||
// After a FAILED wake the only useful next step is fixing the target (wrong MAC,
|
||||
// host moved NIC, command gone) — otherwise a configured-but-broken host would be
|
||||
// stuck behind a button that keeps failing with no way to edit it.
|
||||
const offerConfig = !hasTarget || Boolean(state.error);
|
||||
action.textContent = state.waking ? 'Waking …' : offerConfig ? 'Configure WoL' : 'Wake';
|
||||
action.disabled = state.waking;
|
||||
},
|
||||
|
||||
/** Banner button: wake the host, or open the setup dialog when nothing is configured. */
|
||||
hostWakeAction() {
|
||||
const state = this._hostWake;
|
||||
if (!state || !state.sessionId || state.waking) return;
|
||||
if (state.wakeConfigured === 'none' || state.error) {
|
||||
this.openWakeConfigDialog();
|
||||
return;
|
||||
}
|
||||
this.wakeRemoteHost();
|
||||
},
|
||||
|
||||
/** POST the manual wake for the active session and follow the result. */
|
||||
async wakeRemoteHost() {
|
||||
const state = this._hostWake;
|
||||
if (!state || !state.sessionId) return;
|
||||
const sessionId = state.sessionId;
|
||||
state.waking = true;
|
||||
// The button path holds nothing: whatever the user typed went into the stalled pane
|
||||
// over the WebSocket and is gone. Saying otherwise is a promise the next keystroke
|
||||
// disproves.
|
||||
state.queuedInput = false;
|
||||
state.error = '';
|
||||
this._renderHostWakeBanner();
|
||||
try {
|
||||
const res = await fetch(`/api/sessions/${encodeURIComponent(sessionId)}/wake`, { method: 'POST' });
|
||||
const data = await res.json();
|
||||
if (this._hostWake !== state || state.sessionId !== sessionId) return;
|
||||
state.waking = false;
|
||||
if (!data.success) {
|
||||
// The ROUTE is the authority on whether a target is configured, so ask it again
|
||||
// (`/reachability` reports `wakeConfigured`) rather than pattern-matching the
|
||||
// error message: the message is prose, and the code is generic (`INVALID_INPUT`
|
||||
// covers "Not a remote session" too).
|
||||
state.error = data.error || 'Wake failed';
|
||||
this._renderHostWakeBanner();
|
||||
await this._pollHostReachability(true);
|
||||
return;
|
||||
}
|
||||
state.reachable = data.data.reachable !== false;
|
||||
state.wakeConfigured = data.data.wakeConfigured || state.wakeConfigured;
|
||||
if (state.reachable) {
|
||||
this.showToast(`${state.label || 'Remote host'} is awake`, 'success');
|
||||
} else {
|
||||
state.error = 'timeout';
|
||||
}
|
||||
this._renderHostWakeBanner();
|
||||
} catch (err) {
|
||||
if (this._hostWake !== state) return;
|
||||
state.waking = false;
|
||||
state.error = err && err.message ? err.message : 'Wake failed';
|
||||
this._renderHostWakeBanner();
|
||||
}
|
||||
},
|
||||
|
||||
/**
|
||||
* Why the host could not be read. In multi-user mode `GET /api/remote-hosts` returns
|
||||
* `[]` to a non-admin, so "Remote host not found" would blame a config the user simply
|
||||
* is not allowed to see — the save is admin-only, and that is what it should say.
|
||||
*/
|
||||
_wakeConfigUnavailableMessage() {
|
||||
const me = window.__codemanUser || {};
|
||||
return me.multiUser && me.role !== 'admin' ? 'Wake-on-LAN configuration is admin-only' : 'Remote host not found';
|
||||
},
|
||||
|
||||
/** Open the small WoL dialog for the banner's host, pre-filled from the host config. */
|
||||
async openWakeConfigDialog() {
|
||||
const state = this._hostWake;
|
||||
const session = state && state.sessionId && this.sessions ? this.sessions.get(state.sessionId) : null;
|
||||
if (!session || !session.remote) return;
|
||||
const hostId = session.remote.hostId;
|
||||
const label = this.$('wakeConfigHostLabel');
|
||||
const mac = this.$('wakeConfigMac');
|
||||
const command = this.$('wakeConfigCommand');
|
||||
const status = this.$('wakeConfigStatus');
|
||||
if (!mac || !command) return;
|
||||
|
||||
mac.value = session.remote.wakeMac || '';
|
||||
command.value = session.remote.wakeCommand || '';
|
||||
if (label) label.textContent = session.remote.label || hostId;
|
||||
if (status) status.textContent = '';
|
||||
this._wakeConfigHostId = hostId;
|
||||
const modal = this.$('wakeConfigModal');
|
||||
if (modal) modal.classList.add('active');
|
||||
|
||||
// Read the saved host so the dialog shows what is actually persisted (the session
|
||||
// payload may predate a change made in another tab).
|
||||
try {
|
||||
const res = await fetch('/api/remote-hosts');
|
||||
const data = await res.json();
|
||||
const hosts = data.success ? data.data : [];
|
||||
const host = Array.isArray(hosts) ? hosts.find((item) => item.id === hostId) : null;
|
||||
if (host && this._wakeConfigHostId === hostId) {
|
||||
mac.value = host.wakeMac || '';
|
||||
command.value = host.wakeCommand || '';
|
||||
} else if (!host && this._wakeConfigHostId === hostId && status) {
|
||||
// Say it up front rather than only when Save fails.
|
||||
status.textContent = this._wakeConfigUnavailableMessage();
|
||||
}
|
||||
} catch {
|
||||
/* The form is already usable from the session payload. */
|
||||
}
|
||||
},
|
||||
|
||||
closeWakeConfigDialog() {
|
||||
const modal = this.$('wakeConfigModal');
|
||||
if (modal) modal.classList.remove('active');
|
||||
this._wakeConfigHostId = null;
|
||||
},
|
||||
|
||||
/** Save MAC/command for the host, then re-check whether the session can wake now. */
|
||||
async saveWakeConfig() {
|
||||
const hostId = this._wakeConfigHostId;
|
||||
const mac = this.$('wakeConfigMac');
|
||||
const command = this.$('wakeConfigCommand');
|
||||
const status = this.$('wakeConfigStatus');
|
||||
const save = this.$('wakeConfigSave');
|
||||
if (!hostId || !mac || !command) return;
|
||||
|
||||
const macValue = mac.value.trim();
|
||||
const commandValue = command.value.trim();
|
||||
if (
|
||||
macValue &&
|
||||
!/^[0-9a-fA-F]{2}([:-][0-9a-fA-F]{2}){5}(\s*,\s*[0-9a-fA-F]{2}([:-][0-9a-fA-F]{2}){5})*$/.test(macValue)
|
||||
) {
|
||||
if (status) status.textContent = 'MAC must look like 04:d9:f5:80:c6:58 (comma-separated for several).';
|
||||
return;
|
||||
}
|
||||
if (commandValue && /\s/.test(commandValue)) {
|
||||
if (status) status.textContent = 'The wake command must be a single executable path (no arguments).';
|
||||
return;
|
||||
}
|
||||
|
||||
if (save) save.disabled = true;
|
||||
if (status) status.textContent = 'Saving …';
|
||||
try {
|
||||
const listRes = await fetch('/api/remote-hosts');
|
||||
const listData = await listRes.json();
|
||||
const hosts = listData.success ? listData.data : [];
|
||||
const host = Array.isArray(hosts) ? hosts.find((item) => item.id === hostId) : null;
|
||||
if (!host) throw new Error(this._wakeConfigUnavailableMessage());
|
||||
// PUT takes the whole host (schema-validated), so send back everything we know and
|
||||
// only replace the wake fields. `undefined` drops the key entirely.
|
||||
const payload = {
|
||||
...host,
|
||||
wakeMac: macValue || undefined,
|
||||
wakeCommand: commandValue || undefined,
|
||||
};
|
||||
const res = await fetch(`/api/remote-hosts/${encodeURIComponent(hostId)}`, {
|
||||
method: 'PUT',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify(payload),
|
||||
});
|
||||
const data = await res.json();
|
||||
if (!data.success) throw new Error(data.error || 'Save failed');
|
||||
this.showToast('Wake settings saved', 'success');
|
||||
this.closeWakeConfigDialog();
|
||||
// The server re-resolves host config for live sessions, so the banner can offer
|
||||
// the wake right away — probe fresh instead of waiting out the poll interval.
|
||||
await this._pollHostReachability(true);
|
||||
} catch (err) {
|
||||
if (status) status.textContent = err && err.message ? err.message : 'Save failed';
|
||||
} finally {
|
||||
if (save) save.disabled = false;
|
||||
}
|
||||
},
|
||||
|
||||
/**
|
||||
* SSE `remote:hostWaking` — a wake is running (ours or one started by typing).
|
||||
*
|
||||
* ⚠️ The ONLY definition of this handler: `panels-ui.js` must not define it too.
|
||||
* Both mix into `Codeman.prototype` and this file loads later, so a second copy
|
||||
* would be silently shadowed (the guard in `sse-dispatch-table.test.ts` sees that a
|
||||
* handler exists, not that two modules claim the same name). The toast is
|
||||
* deliberately UNCONDITIONAL — a wake can start for a background session (input on
|
||||
* a non-active tab) where there is no banner to update.
|
||||
*/
|
||||
_onRemoteHostWaking(data) {
|
||||
const label = data && data.label ? data.label : 'Remote host';
|
||||
// A create-path wake (the user pressed Run / Attach) has no session yet, so
|
||||
// nothing is queued behind it — the wording has to say what actually happens.
|
||||
const forNewSession = Boolean(data && data.forNewSession);
|
||||
// Only the typing path buffers bytes; the wake button and the send-and-wait path
|
||||
// hold none, and a browser keystroke never reaches the registry at all.
|
||||
const queuedInput = Boolean(data && data.queuedInput);
|
||||
// Long enough to cover the wake + attach (~10s measured on a warm S3), and it
|
||||
// is replaced by `remote:sessionReconnected` the moment the pane is back.
|
||||
this.showToast(
|
||||
forNewSession
|
||||
? `Waking ${label} … the session starts when it is back`
|
||||
: queuedInput
|
||||
? `Waking ${label} … input is queued`
|
||||
: `Waking ${label} … waiting for it to come back`,
|
||||
'info',
|
||||
{ duration: 12000 }
|
||||
);
|
||||
const state = this._hostWake;
|
||||
if (!state || !data || state.sessionId !== data.sessionId) return;
|
||||
state.waking = true;
|
||||
state.queuedInput = queuedInput;
|
||||
state.error = '';
|
||||
if (data.label) state.label = data.label;
|
||||
this._renderHostWakeBanner();
|
||||
},
|
||||
|
||||
/** SSE `remote:hostWakeFailed` — the host did not come back in time. */
|
||||
_onRemoteHostWakeFailed(data) {
|
||||
const label = data && data.label ? data.label : 'Remote host';
|
||||
const forNewSession = Boolean(data && data.forNewSession);
|
||||
const queuedInput = Boolean(data && data.queuedInput);
|
||||
this.showToast(
|
||||
forNewSession
|
||||
? `${label} did not wake up — no session was started`
|
||||
: queuedInput
|
||||
? `${label} did not wake up — queued input is still held`
|
||||
: `${label} did not wake up`,
|
||||
'error',
|
||||
{ duration: 15000 }
|
||||
);
|
||||
const state = this._hostWake;
|
||||
if (!state || !data || state.sessionId !== data.sessionId) return;
|
||||
state.waking = false;
|
||||
state.queuedInput = queuedInput;
|
||||
state.error = 'timeout';
|
||||
state.reachable = false;
|
||||
this._renderHostWakeBanner();
|
||||
},
|
||||
});
|
||||
@@ -252,6 +252,7 @@
|
||||
'Ultracode Agents': 'Ultracode 智能体',
|
||||
'Ultracode Floating Windows': 'Ultracode 浮动窗口',
|
||||
'Approvals Inbox': '审批收件箱',
|
||||
'Auto-name Sessions': '自动命名会话',
|
||||
Approvals: '审批',
|
||||
'Prompts waiting on you, across all sessions': '所有会话中等待您处理的提示',
|
||||
'No pending approvals': '没有待处理的审批',
|
||||
@@ -285,6 +286,34 @@
|
||||
'Prompt sent': '提示已发送',
|
||||
'Inserted, press Enter in the terminal to send': '已插入,在终端中按 Enter 发送',
|
||||
'Could not reach the session': '无法连接到会话',
|
||||
'Custom model endpoints': '自定义模型端点',
|
||||
'Point a harness at your own OpenAI-compatible server (llama.cpp, vLLM, DGX Spark, Azure AI Foundry, OpenRouter) instead of its native cloud backend. When on, the Run menu offers an extra entry per harness that supports it, per saved endpoint.':
|
||||
'让工具指向您自己的兼容 OpenAI 服务器(llama.cpp、vLLM、DGX Spark、Azure AI Foundry、OpenRouter),而非其原生云端后端。开启后,"运行"菜单会为每个支持此功能的工具、每个已保存的端点新增一个条目。',
|
||||
'Enable custom model endpoints': '启用自定义模型端点',
|
||||
'Adds a per-endpoint entry to the Run menu for every harness that can redirect to one.':
|
||||
'为每个可重定向到端点的工具,在"运行"菜单中添加对应条目。',
|
||||
'No endpoints yet. Add one below to point a harness at a local or cloud OpenAI-compatible server.':
|
||||
'暂无端点。请在下方添加一个,以便将工具指向本地或云端的兼容 OpenAI 服务器。',
|
||||
Discover: '发现模型',
|
||||
'+ Add endpoint': '+ 添加端点',
|
||||
'Add endpoint': '添加端点',
|
||||
Id: 'ID',
|
||||
'Short, stable — used in URLs, never shown to the CLI.': '简短且固定 — 用于 URL,不会展示给 CLI。',
|
||||
Label: '标签',
|
||||
'Base URL': '基础 URL',
|
||||
'API key': 'API 密钥',
|
||||
'Optional. Left blank on edit keeps the existing key.': '可选。编辑时留空将保留现有密钥。',
|
||||
'Auth header': '认证请求头',
|
||||
'Never send both — some servers hang indefinitely.': '切勿同时发送两者 — 部分服务器会因此无限期挂起。',
|
||||
'Authorization: Bearer (default)': 'Authorization: Bearer(默认)',
|
||||
'api-key header (Azure)': 'api-key 请求头(Azure)',
|
||||
'Default model': '默认模型',
|
||||
'What the Run-menu picker applies for this endpoint. Discover models first.':
|
||||
'运行菜单选择器会为此端点应用该模型。请先发现可用模型。',
|
||||
'Custom Endpoints': '自定义端点',
|
||||
'Choose a model': '选择模型',
|
||||
'That endpoint no longer exists': '该端点已不存在',
|
||||
'No models discovered for this endpoint yet': '此端点尚未发现任何模型',
|
||||
'Subagent Options': '子智能体选项',
|
||||
'Enable Tracking': '启用跟踪',
|
||||
'Active Tab Only': '仅活动标签页',
|
||||
@@ -520,6 +549,7 @@
|
||||
'Respawn Blocked': '重生已阻止',
|
||||
'Task Complete': '任务完成',
|
||||
'Copied to clipboard': '已复制到剪贴板',
|
||||
'Nothing to copy': '没有可复制的内容',
|
||||
// Terminal touch-selection bar (long-press to select). The bar is a sibling of
|
||||
// `.xterm`, not a descendant, so SKIP_SELECTOR does not cover it and these apply.
|
||||
Copy: '复制',
|
||||
|
||||
@@ -213,6 +213,36 @@
|
||||
<button class="offline-banner-retry" id="offlineBannerRetry" onclick="app.retryConnection()">Retry now</button>
|
||||
</div>
|
||||
|
||||
<!-- Remote-host unreachable: the machine SLEEPS, the local ssh pane stalls
|
||||
silently (send-keys succeeds against it, so typed input would vanish) and
|
||||
Codeman can wake it. Amber, not red: the session is fine, the host is
|
||||
asleep. Without a configured wake target the action becomes "Configure
|
||||
WoL" and opens the small config dialog. -->
|
||||
<div class="offline-banner host-wake-banner" id="hostWakeBanner" role="status" hidden>
|
||||
<span class="offline-banner-dot" aria-hidden="true"></span>
|
||||
<span class="offline-banner-text" id="hostWakeBannerText">Remote host is unreachable</span>
|
||||
<span class="offline-banner-detail" id="hostWakeBannerDetail"></span>
|
||||
<button class="offline-banner-retry" id="hostWakeBannerAction" onclick="app.hostWakeAction()">Wake</button>
|
||||
</div>
|
||||
|
||||
<!-- Reboot-restore offer: shown when the server found sessions a host reboot
|
||||
killed and is asking whether to rebuild them. Populated by
|
||||
reboot-restore-ui.js; nothing is created until the user clicks. -->
|
||||
<div class="reboot-restore-banner" id="rebootRestoreBanner" role="status" hidden>
|
||||
<span class="reboot-restore-banner-icon" aria-hidden="true">↺</span>
|
||||
<span class="reboot-restore-banner-text" id="rebootRestoreBannerText"></span>
|
||||
<span class="reboot-restore-banner-detail" id="rebootRestoreBannerDetail"></span>
|
||||
<span class="reboot-restore-banner-note">Conversations return; terminal history does not.</span>
|
||||
<button
|
||||
class="reboot-restore-banner-accept"
|
||||
id="rebootRestoreBannerAccept"
|
||||
onclick="app.restoreRebootSessions()"
|
||||
>
|
||||
Restore
|
||||
</button>
|
||||
<button class="reboot-restore-banner-dismiss" onclick="app.dismissRebootRestore()">Dismiss</button>
|
||||
</div>
|
||||
|
||||
<!-- Timer Banner (shown when timed run is active) -->
|
||||
<div class="timer-banner" id="timerBanner" style="display: none;">
|
||||
<div class="timer-content">
|
||||
@@ -650,6 +680,14 @@
|
||||
<button class="run-mode-option" data-mode="omp" onclick="app.setRunMode('omp')">
|
||||
<span class="run-mode-dot omp"></span>OMP
|
||||
</button>
|
||||
<!-- Custom Model Endpoint Profiles (docs/custom-model-endpoints-plan.md): one
|
||||
generated entry per (harness, saved endpoint) pair, e.g. "Claude Code
|
||||
(llama.cpp)". Built entirely by _refreshCustomModelRunOptions() — hidden
|
||||
when the feature is off or no endpoint has a usable default model, never
|
||||
a fixed per-harness duplicate in this markup. -->
|
||||
<div class="run-mode-sep" id="runModeCustomModelSep" style="display: none;"></div>
|
||||
<div class="run-mode-header" id="runModeCustomModelHeader" style="display: none;">Custom Endpoints</div>
|
||||
<div class="run-mode-custom-models" id="runModeCustomModels"></div>
|
||||
<div class="run-mode-sep"></div>
|
||||
<button class="run-mode-option" data-mode="shell" onclick="app.setRunMode('shell')">
|
||||
<span class="run-mode-dot shell"></span>Terminal / Shell
|
||||
@@ -902,6 +940,67 @@
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Custom Model Endpoint Profiles: "which model" picker (docs/custom-model-endpoints-plan.md).
|
||||
Shown only when the chosen endpoint has more than one discovered model — see
|
||||
selectCustomModelEntry() in session-ui.js, which skips straight to launch otherwise. -->
|
||||
<div class="modal" id="customModelPickModal">
|
||||
<div class="modal-backdrop" onclick="app.closeCustomModelPickModal()"></div>
|
||||
<div class="modal-content modal-sm">
|
||||
<div class="modal-header">
|
||||
<h3 id="customModelPickTitle">Choose a model</h3>
|
||||
<button class="modal-close" onclick="app.closeCustomModelPickModal()" aria-label="Close model picker">×</button>
|
||||
</div>
|
||||
<div class="modal-body">
|
||||
<p class="form-hint" id="customModelPickHint"></p>
|
||||
<div id="customModelPickList" class="run-mode-custom-models"></div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Custom Model Endpoint Profiles: llama-swap model-swap confirmation
|
||||
(docs/custom-model-endpoints-plan.md) — replaces a native confirm()
|
||||
popup, shown when switching would unload a model another live
|
||||
session is actively using. See _confirmModelSwap() in session-ui.js. -->
|
||||
<div class="modal" id="customModelSwapConfirmModal">
|
||||
<div class="modal-backdrop" onclick="app._resolveModelSwapConfirm(false)"></div>
|
||||
<div class="modal-content modal-sm">
|
||||
<div class="modal-header">
|
||||
<h3>Switch models?</h3>
|
||||
<button class="modal-close" onclick="app._resolveModelSwapConfirm(false)" aria-label="Cancel">×</button>
|
||||
</div>
|
||||
<div class="modal-body">
|
||||
<p class="form-hint" id="customModelSwapConfirmMessage"></p>
|
||||
</div>
|
||||
<div class="modal-footer">
|
||||
<button class="btn-toolbar" onclick="app._resolveModelSwapConfirm(false)">Cancel</button>
|
||||
<button class="btn-toolbar btn-primary" onclick="app._resolveModelSwapConfirm(true)">Switch anyway</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Custom Model Endpoint Profiles: context-window-too-small warning
|
||||
(docs/custom-model-endpoints-plan.md) — shown before launching a CLI
|
||||
whose own fixed system-prompt/tool-schema overhead exceeds the
|
||||
model's real discovered context, which guarantees a first-message
|
||||
failure regardless of CLAUDE_CODE_MAX_CONTEXT_TOKENS. See
|
||||
_confirmContextWarning() in session-ui.js. -->
|
||||
<div class="modal" id="customModelContextWarningModal">
|
||||
<div class="modal-backdrop" onclick="app._resolveContextWarningConfirm(false)"></div>
|
||||
<div class="modal-content modal-sm">
|
||||
<div class="modal-header">
|
||||
<h3>Context window too small</h3>
|
||||
<button class="modal-close" onclick="app._resolveContextWarningConfirm(false)" aria-label="Cancel">×</button>
|
||||
</div>
|
||||
<div class="modal-body">
|
||||
<p class="form-hint" id="customModelContextWarningMessage" style="white-space: pre-wrap;"></p>
|
||||
</div>
|
||||
<div class="modal-footer">
|
||||
<button class="btn-toolbar" onclick="app._resolveContextWarningConfirm(false)">Cancel</button>
|
||||
<button class="btn-toolbar btn-primary" onclick="app._resolveContextWarningConfirm(true)">Launch anyway</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Cron Jobs Modal -->
|
||||
<div class="modal" id="cronModal">
|
||||
<div class="modal-backdrop" onclick="app.closeCron()"></div>
|
||||
@@ -2047,6 +2146,13 @@
|
||||
</div>
|
||||
<label class="switch switch-sm"><input type="checkbox" id="appSettingsLineageLines" checked><span class="slider"></span></label>
|
||||
</div>
|
||||
<div class="set-row" id="appSettingsAutoNameSessionsItem" data-search="auto name session title first prompt tab rename">
|
||||
<div class="set-row-text">
|
||||
<span class="set-row-label">Auto-name Sessions <span class="set-tag">synced</span></span>
|
||||
<span class="set-row-desc">Title a new tab after its first prompt, keeping the case prefix. Renamed tabs are never touched.</span>
|
||||
</div>
|
||||
<label class="switch switch-sm"><input type="checkbox" id="appSettingsAutoNameSessions"><span class="slider"></span></label>
|
||||
</div>
|
||||
<div class="set-row" id="appSettingsMobileOverviewItem" data-search="overview home screen phone logo">
|
||||
<div class="set-row-text">
|
||||
<span class="set-row-label">Overview Home Screen <span class="set-tag">phone</span></span>
|
||||
@@ -2191,6 +2297,60 @@
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="set-group" id="customModelEndpointsGroup">
|
||||
<div class="set-group-head"><h4>Custom model endpoints</h4><span class="set-scope">synced</span></div>
|
||||
<p class="set-group-hint">Point a harness at your own OpenAI-compatible server (llama.cpp, vLLM, DGX Spark, Azure AI Foundry, OpenRouter) instead of its native cloud backend. When on, the Run menu offers an extra entry per harness that supports it, per saved endpoint.</p>
|
||||
<div class="set-group-body">
|
||||
<div class="set-row" data-search="custom model endpoint llama.cpp local llm run menu picker">
|
||||
<div class="set-row-text">
|
||||
<span class="set-row-label">Enable custom model endpoints</span>
|
||||
<span class="set-row-desc">Adds a per-endpoint entry to the Run menu for every harness that can redirect to one.</span>
|
||||
</div>
|
||||
<label class="switch switch-sm"><input type="checkbox" id="appSettingsCustomModelEndpoints" onchange="app.applyCustomModelEndpointsVisibility()"><span class="slider"></span></label>
|
||||
</div>
|
||||
<!-- Gated on the toggle above (applyCustomModelEndpointsVisibility): with the
|
||||
feature off, a list of endpoints that do nothing is worse than nothing. -->
|
||||
<div id="customModelEndpointsBody" style="display:none">
|
||||
<div id="customModelHostsList" class="set-group-body" data-search="endpoints"></div>
|
||||
<button type="button" class="btn-toolbar btn-sm" id="customModelHostAddBtn" onclick="app.openCustomModelHostEditor()">+ Add endpoint</button>
|
||||
<div id="customModelHostEditor" class="set-inline-form" style="display:none">
|
||||
<h5 id="customModelHostEditorTitle">Add endpoint</h5>
|
||||
<div class="set-row has-field">
|
||||
<div class="set-row-text"><span class="set-row-label">Id</span><span class="set-row-desc">Short, stable — used in URLs, never shown to the CLI.</span></div>
|
||||
<input type="text" id="customModelHostId" class="set-input" placeholder="llama-cpp-local">
|
||||
</div>
|
||||
<div class="set-row has-field">
|
||||
<div class="set-row-text"><span class="set-row-label">Label</span></div>
|
||||
<input type="text" id="customModelHostLabel" class="set-input" placeholder="llama.cpp (local)">
|
||||
</div>
|
||||
<div class="set-row has-field">
|
||||
<div class="set-row-text"><span class="set-row-label">Base URL</span></div>
|
||||
<input type="text" id="customModelHostBaseUrl" class="set-input" placeholder="http://192.168.1.50:8080">
|
||||
</div>
|
||||
<div class="set-row has-field">
|
||||
<div class="set-row-text"><span class="set-row-label">API key</span><span class="set-row-desc">Optional. Left blank on edit keeps the existing key.</span></div>
|
||||
<input type="password" id="customModelHostApiKey" class="set-input" autocomplete="new-password">
|
||||
</div>
|
||||
<div class="set-row has-field">
|
||||
<div class="set-row-text"><span class="set-row-label">Auth header</span><span class="set-row-desc">Never send both — some servers hang indefinitely.</span></div>
|
||||
<select id="customModelHostAuthStyle" class="set-select">
|
||||
<option value="bearer">Authorization: Bearer (default)</option>
|
||||
<option value="api-key">api-key header (Azure)</option>
|
||||
</select>
|
||||
</div>
|
||||
<div class="set-row has-field">
|
||||
<div class="set-row-text"><span class="set-row-label">Default model</span><span class="set-row-desc">What the Run-menu picker applies for this endpoint. Discover models first.</span></div>
|
||||
<select id="customModelHostDefaultModel" class="set-select" disabled></select>
|
||||
</div>
|
||||
<div class="set-row-actions">
|
||||
<button type="button" class="btn-toolbar btn-sm" onclick="app.saveCustomModelHostFromEditor()">Save</button>
|
||||
<button type="button" class="btn-toolbar btn-sm" onclick="app.closeCustomModelHostEditor()">Cancel</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<!-- ══ Agents & CLIs ═════════════════════════════════════════ -->
|
||||
@@ -2853,6 +3013,11 @@
|
||||
<input type="number" id="remoteHostPort" placeholder="22" min="1" max="65535" autocomplete="off">
|
||||
<span class="form-hint">Optional. Leave blank for the default port 22.</span>
|
||||
</div>
|
||||
<div class="form-row">
|
||||
<label>Wake-on-LAN MAC</label>
|
||||
<input type="text" id="remoteHostWakeMac" placeholder="04:d9:f5:80:c6:58" autocomplete="off" autocapitalize="off" spellcheck="false">
|
||||
<span class="form-hint">Optional. Comma-separated for several NICs. Codeman sends the magic packet itself so a sleeping host can be woken from the session banner.</span>
|
||||
</div>
|
||||
<div class="form-row">
|
||||
<label>Codex Command Override</label>
|
||||
<input type="text" id="remoteHostCodexCommand" placeholder="exec codx personal" autocomplete="off" autocapitalize="off" spellcheck="false">
|
||||
@@ -2861,6 +3026,11 @@
|
||||
<details class="advanced-options">
|
||||
<summary><svg class="set-adv-chev" width="12" height="12" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.4" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true"><path d="M6 9l6 6 6-6"/></svg><span>Advanced SSH</span></summary>
|
||||
<div class="advanced-options-content">
|
||||
<div class="form-row">
|
||||
<label>Wake Command</label>
|
||||
<input type="text" id="remoteHostWakeCommand" placeholder="/home/user/bin/wake-this-host" autocomplete="off" autocapitalize="off" autocorrect="off" spellcheck="false">
|
||||
<span class="form-hint">Optional override for the MAC above (takes precedence). A single executable path, run without a shell — use it when the host needs a router/other machine to send the packet.</span>
|
||||
</div>
|
||||
<div class="form-row">
|
||||
<label>Identity File</label>
|
||||
<input type="text" id="remoteHostIdentityFile" placeholder="~/.ssh/remote_ed25519" autocomplete="off" autocapitalize="off" autocorrect="off" spellcheck="false">
|
||||
@@ -3456,6 +3626,36 @@
|
||||
text is set via value/textContent only: predictor output derives from
|
||||
observable (injectable) content, and the explicit click here is the
|
||||
security boundary (nothing is ever auto-sent). -->
|
||||
<!-- Wake-on-LAN setup for a remote host whose session cannot be woken yet. Kept
|
||||
deliberately small (host is fixed, only the wake fields are editable) so it can
|
||||
be opened from the banner with one click. Persists via PUT /api/remote-hosts/:id. -->
|
||||
<div class="modal" id="wakeConfigModal">
|
||||
<div class="modal-backdrop" onclick="app.closeWakeConfigDialog()"></div>
|
||||
<div class="modal-content">
|
||||
<div class="modal-header">
|
||||
<h3>Wake-on-LAN · <span id="wakeConfigHostLabel"></span></h3>
|
||||
<button class="modal-close" onclick="app.closeWakeConfigDialog()" aria-label="Close">×</button>
|
||||
</div>
|
||||
<div class="modal-body">
|
||||
<div class="form-row">
|
||||
<label>MAC address(es)</label>
|
||||
<input type="text" id="wakeConfigMac" placeholder="04:d9:f5:80:c6:58" autocomplete="off" autocapitalize="off" autocorrect="off" spellcheck="false">
|
||||
<span class="form-hint">Comma-separated for several NICs. Codeman sends the magic packet itself (UDP port 9, broadcast).</span>
|
||||
</div>
|
||||
<div class="form-row">
|
||||
<label>Wake command (optional)</label>
|
||||
<input type="text" id="wakeConfigCommand" placeholder="/home/user/bin/wake-this-host" autocomplete="off" autocapitalize="off" autocorrect="off" spellcheck="false">
|
||||
<span class="form-hint">Takes precedence over the MAC. A single executable path, run without a shell.</span>
|
||||
</div>
|
||||
<div class="form-hint" id="wakeConfigStatus"></div>
|
||||
</div>
|
||||
<div class="modal-footer">
|
||||
<button class="btn-toolbar" onclick="app.closeWakeConfigDialog()">Cancel</button>
|
||||
<button class="btn-toolbar btn-primary" id="wakeConfigSave" onclick="app.saveWakeConfig()">Save</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="modal" id="readMyMindModal">
|
||||
<div class="modal-backdrop" onclick="app.closeReadMyMind()"></div>
|
||||
<div class="modal-content readmymind-modal">
|
||||
@@ -3528,8 +3728,10 @@
|
||||
<script defer src="readmymind-ui.js"></script>
|
||||
<script defer src="ultracode-panel.js"></script>
|
||||
<script defer src="approvals-ui.js"></script>
|
||||
<script defer src="reboot-restore-ui.js"></script>
|
||||
<script defer src="admin-ui.js"></script>
|
||||
<script defer src="session-ui.js"></script>
|
||||
<script defer src="host-wake-ui.js"></script>
|
||||
<script defer src="webview-tabs.js"></script>
|
||||
<script defer src="mobile-overview.js"></script>
|
||||
<script defer src="home-sessions.js"></script>
|
||||
|
||||
@@ -3241,6 +3241,43 @@ html:is([data-skin="paper-gray"], [data-skin="solarized-light"], [data-skin="cat
|
||||
other banners. The overlay is fixed and handles its own insets.
|
||||
============================================================================ */
|
||||
@media (max-width: 599px) {
|
||||
/* Reboot-restore banner: the same treatment as the offline banner below. Its
|
||||
text and note are nowrap and the two buttons cannot shrink, so without this
|
||||
the actions are pushed off a phone-width viewport and become unreachable. */
|
||||
.reboot-restore-banner {
|
||||
padding: 0.4rem 0.5rem;
|
||||
padding-left: calc(0.5rem + var(--safe-area-left));
|
||||
padding-right: calc(0.5rem + var(--safe-area-right));
|
||||
font-size: 0.7rem;
|
||||
gap: 0.4rem;
|
||||
}
|
||||
|
||||
/* The session names and the scrollback note are the first things to go. The
|
||||
count plus the two buttons carry the message on their own, and the note
|
||||
survives as the accept button's title. */
|
||||
.reboot-restore-banner-detail,
|
||||
.reboot-restore-banner-note {
|
||||
display: none;
|
||||
}
|
||||
|
||||
/* A flex item will not shrink below its content width at the default
|
||||
`min-width: auto`, so without this the nowrap text pushes the buttons off a
|
||||
360px viewport and the ellipsis never engages. */
|
||||
.reboot-restore-banner-text {
|
||||
min-width: 0;
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
}
|
||||
|
||||
.reboot-restore-banner-accept,
|
||||
.reboot-restore-banner-dismiss {
|
||||
padding: 0.25rem 0.5rem;
|
||||
}
|
||||
|
||||
.reboot-restore-banner-accept {
|
||||
margin-left: auto;
|
||||
}
|
||||
|
||||
.offline-banner {
|
||||
padding: 0.4rem 0.5rem;
|
||||
padding-left: calc(0.5rem + var(--safe-area-left));
|
||||
|
||||
+138
-4
@@ -92,6 +92,9 @@ Object.assign(CodemanApp.prototype, {
|
||||
_onRemoteSessionReconnected(data) {
|
||||
const id = this.getShortId(data.sessionId);
|
||||
this.showToast(`Remote session ${id} reconnected`, 'success');
|
||||
// A successful reattach (the wake flow's own, or the watcher's) means the host is
|
||||
// back: drop the "unreachable" banner without waiting out the poll interval.
|
||||
if (this.activeSessionId === data.sessionId) this._pollHostReachability?.(true);
|
||||
},
|
||||
|
||||
_onRemoteReconnectExhausted(data) {
|
||||
@@ -116,6 +119,16 @@ Object.assign(CodemanApp.prototype, {
|
||||
},
|
||||
|
||||
|
||||
// Wake-on-LAN from user input on a sleeping remote host (see remote-wake.ts).
|
||||
// ⚠️ The `remote:hostWaking` / `remote:hostWakeFailed` HANDLERS live in
|
||||
// `host-wake-ui.js`, which owns the banner state. They are NOT redefined here:
|
||||
// both files mix into `CodemanApp.prototype` and `host-wake-ui.js` is loaded
|
||||
// later, so a second definition would silently shadow the banner update (and the
|
||||
// toast would never fire — the exact silent no-op `sse-dispatch-table.test.ts`
|
||||
// exists to prevent, which cannot see shadowing). The toasts are shown from the
|
||||
// host-wake-ui handlers instead.
|
||||
|
||||
|
||||
// Bash tools
|
||||
_onBashToolStart(data) {
|
||||
this.handleBashToolStart(data.sessionId, data.tool);
|
||||
@@ -5484,12 +5497,25 @@ Object.assign(CodemanApp.prototype, {
|
||||
return this.showToast(message, type);
|
||||
},
|
||||
|
||||
/**
|
||||
* `duration` defaults to 3000ms for every toast type. A message worth
|
||||
* reading rather than glancing at (e.g. "Session started on the native
|
||||
* backend — could not apply the custom endpoint: <the actual reason>")
|
||||
* passes an explicit `opts.duration: 0` at its own call site instead of
|
||||
* widening the default: this used to default every `error` toast to
|
||||
* sticky, and with no cap on `.toast-container` and no eviction, a
|
||||
* repeatedly failing path (a flapping SSE reconnect, a poll loop) stacked
|
||||
* sticky toasts off the bottom of the viewport where they could not be
|
||||
* read or dismissed. Every toast still gets an explicit close button
|
||||
* regardless of duration.
|
||||
*/
|
||||
showToast(message, type = 'info', opts = {}) {
|
||||
const { duration = 3000, action } = opts;
|
||||
const toast = document.createElement('div');
|
||||
toast.className = `toast toast-${type}`;
|
||||
|
||||
const msgSpan = document.createElement('span');
|
||||
msgSpan.className = 'toast-message';
|
||||
msgSpan.textContent = message;
|
||||
toast.appendChild(msgSpan);
|
||||
|
||||
@@ -5501,6 +5527,20 @@ Object.assign(CodemanApp.prototype, {
|
||||
toast.appendChild(btn);
|
||||
}
|
||||
|
||||
let dismissTimer = null;
|
||||
const dismiss = () => {
|
||||
if (dismissTimer) clearTimeout(dismissTimer);
|
||||
toast.classList.remove('show');
|
||||
setTimeout(() => toast.remove(), 200);
|
||||
};
|
||||
|
||||
const closeBtn = document.createElement('button');
|
||||
closeBtn.className = 'toast-close';
|
||||
closeBtn.textContent = '×';
|
||||
closeBtn.setAttribute('aria-label', 'Dismiss');
|
||||
closeBtn.onclick = (e) => { e.stopPropagation(); dismiss(); };
|
||||
toast.appendChild(closeBtn);
|
||||
|
||||
// Cache toast container reference
|
||||
if (!this._toastContainer) {
|
||||
this._toastContainer = document.querySelector('.toast-container');
|
||||
@@ -5514,10 +5554,104 @@ Object.assign(CodemanApp.prototype, {
|
||||
|
||||
requestAnimationFrame(() => toast.classList.add('show'));
|
||||
|
||||
setTimeout(() => {
|
||||
toast.classList.remove('show');
|
||||
setTimeout(() => toast.remove(), 200);
|
||||
}, duration);
|
||||
if (duration > 0) {
|
||||
dismissTimer = setTimeout(dismiss, duration);
|
||||
}
|
||||
|
||||
// Most callers ignore this — a handle exists for a long-running toast a caller needs
|
||||
// to update or dismiss itself once its own condition resolves (e.g. a "loading model"
|
||||
// toast a poll loop dismisses once the model reports ready).
|
||||
return { dismiss, setMessage: (text) => { msgSpan.textContent = text; } };
|
||||
},
|
||||
|
||||
/**
|
||||
* A prominent, screen-centred status banner — for the small set of messages that are
|
||||
* genuinely worth interrupting the eye for rather than living in the corner with every
|
||||
* other toast (currently: a custom-model session's "switching backends" and "loading
|
||||
* model" states, both of which can sit on screen for well over a minute and are easy to
|
||||
* mistake for nothing happening). Non-blocking (`pointer-events: none` on the wrapper,
|
||||
* restored only on the card) — an info banner is never a gate the user has to dismiss to
|
||||
* keep working. Only one is ever shown at a time (the DOM node is created once and
|
||||
* reused), which matches every current caller: each hands off to the next rather than
|
||||
* stacking.
|
||||
*
|
||||
* `opts.type` — `'info'` (default, spinner, no close button — a caller ends it itself via
|
||||
* `dismiss()`) or `'error'` (no spinner — nothing is in progress once this shows — with a
|
||||
* close button, since a sticky error the user cannot dismiss would just sit there). The
|
||||
* DOM is rebuilt fresh each call rather than patched, since which children exist differs
|
||||
* by type; `setMessage` still only ever touches the text node afterwards.
|
||||
*
|
||||
* `opts.onCancel` — when given (any type, but in practice only 'info': an 'error' banner
|
||||
* already has its own close button), renders a "Cancel" button that calls it on click.
|
||||
* The callback owns everything that follows (dismissing the banner, stopping whatever
|
||||
* loop this was showing progress for, closing a session it was for) — this helper only
|
||||
* renders the button and wires the click, the same "caller decides what cancel means"
|
||||
* split as `_confirmModelSwap`'s promise-resolving buttons.
|
||||
*/
|
||||
_showCenterStatus(message, opts = {}) {
|
||||
const { type = 'info', onCancel } = opts;
|
||||
let el = document.getElementById('customModelCenterStatus');
|
||||
if (!el) {
|
||||
el = document.createElement('div');
|
||||
el.id = 'customModelCenterStatus';
|
||||
document.body.appendChild(el);
|
||||
}
|
||||
// A pending hide from a PREVIOUS dismiss() (e.g. switchingToast.dismiss() right
|
||||
// before this same-origin call reopens the banner within its 200ms fade) must
|
||||
// never fire against the node this call is about to show — clear it before
|
||||
// reusing the shared DOM node, or the old timer hides the fresh banner ~200ms in.
|
||||
if (el._hideTimer) {
|
||||
clearTimeout(el._hideTimer);
|
||||
el._hideTimer = null;
|
||||
}
|
||||
el.className = `center-status-banner center-status-${type}`;
|
||||
el.innerHTML = '';
|
||||
const dismiss = () => {
|
||||
el.classList.remove('show');
|
||||
el._hideTimer = setTimeout(() => {
|
||||
el.hidden = true;
|
||||
el._hideTimer = null;
|
||||
}, 200);
|
||||
};
|
||||
if (type !== 'error') {
|
||||
const spinner = document.createElement('span');
|
||||
spinner.className = 'center-status-spinner';
|
||||
spinner.setAttribute('aria-hidden', 'true');
|
||||
el.appendChild(spinner);
|
||||
}
|
||||
const text = document.createElement('span');
|
||||
text.className = 'center-status-text';
|
||||
text.textContent = message;
|
||||
el.appendChild(text);
|
||||
if (type === 'error') {
|
||||
const closeBtn = document.createElement('button');
|
||||
closeBtn.className = 'center-status-close';
|
||||
closeBtn.textContent = '×';
|
||||
closeBtn.setAttribute('aria-label', 'Dismiss');
|
||||
closeBtn.onclick = (e) => {
|
||||
e.stopPropagation();
|
||||
dismiss();
|
||||
};
|
||||
el.appendChild(closeBtn);
|
||||
} else if (onCancel) {
|
||||
const cancelBtn = document.createElement('button');
|
||||
cancelBtn.className = 'center-status-cancel';
|
||||
cancelBtn.textContent = 'Cancel';
|
||||
cancelBtn.onclick = (e) => {
|
||||
e.stopPropagation();
|
||||
onCancel();
|
||||
};
|
||||
el.appendChild(cancelBtn);
|
||||
}
|
||||
el.hidden = false;
|
||||
requestAnimationFrame(() => el.classList.add('show'));
|
||||
return {
|
||||
dismiss,
|
||||
setMessage: (next) => {
|
||||
const t = el.querySelector('.center-status-text');
|
||||
if (t) t.textContent = next;
|
||||
},
|
||||
};
|
||||
},
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,142 @@
|
||||
/**
|
||||
* @fileoverview Reboot-restore banner: offer back the sessions a host reboot destroyed.
|
||||
*
|
||||
* A host reboot takes the tmux server down with it, so every session's pane dies
|
||||
* and the board comes up empty. The server works out what was running from the
|
||||
* records it still holds at boot, and this banner asks the user whether to
|
||||
* rebuild them. Nothing is created until they click, because the server's
|
||||
* reboot guess is a heuristic and a wrong automatic restore would spawn CLI
|
||||
* processes nobody asked for.
|
||||
*
|
||||
* Seeded from `GET /api/reboot-restore` on init and again on every SSE reconnect,
|
||||
* because the tab most likely to want this is one that was open across the reboot
|
||||
* and reconnects to a server that came back up with an empty board. Restore posts to
|
||||
* `POST /api/reboot-restore/restore` and Dismiss posts to
|
||||
* `POST /api/reboot-restore/dismiss`. Dismiss always clears the banner; Restore
|
||||
* re-reads the plan afterwards, because the server puts back anything it could
|
||||
* not build for a reason that may pass, such as a session limit or an agent that
|
||||
* would not start. The restored sessions arrive as ordinary `session:created`
|
||||
* events, so no extra rendering is needed here.
|
||||
*
|
||||
* The banner says that terminal history did not survive, because a restored
|
||||
* session is a new pane: the conversation continues and the scrollback does not.
|
||||
* Saying so is what keeps an empty pane from reading as a broken restore.
|
||||
* Backend: src/web/reboot-restore-registry.ts, src/web/routes/reboot-restore-routes.ts.
|
||||
*
|
||||
* @mixin Extends CodemanApp.prototype via Object.assign
|
||||
* @dependency app.js (CodemanApp class, showToast)
|
||||
* @dependency api-client.js at runtime (this._api / this._apiJson)
|
||||
* @loadorder 11.65, after approvals-ui.js and before admin-ui.js (11.7)
|
||||
*/
|
||||
|
||||
/** Plain-language wording for one skip reason, for the toast after a restore. */
|
||||
function rebootSkipReason(reason) {
|
||||
switch (reason) {
|
||||
case 'workspace-missing':
|
||||
return 'workspace is gone';
|
||||
case 'workspace-forbidden':
|
||||
return 'workspace is outside your space';
|
||||
case 'already-live':
|
||||
return 'already open';
|
||||
case 'capacity-reached':
|
||||
return 'session limit reached';
|
||||
case 'rebuild-failed':
|
||||
return 'the agent would not start';
|
||||
default:
|
||||
return reason;
|
||||
}
|
||||
}
|
||||
|
||||
Object.assign(CodemanApp.prototype, {
|
||||
/** Ask the server whether a reboot left anything on offer, and show the banner if so. */
|
||||
async initRebootRestoreBanner() {
|
||||
const data = await this._apiJson('/api/reboot-restore');
|
||||
const sessions = data?.sessions ?? [];
|
||||
if (sessions.length === 0) return;
|
||||
this._rebootRestoreSessions = sessions;
|
||||
this.renderRebootRestoreBanner();
|
||||
},
|
||||
|
||||
renderRebootRestoreBanner() {
|
||||
const banner = this.$('rebootRestoreBanner');
|
||||
if (!banner) return;
|
||||
const sessions = this._rebootRestoreSessions ?? [];
|
||||
if (sessions.length === 0) {
|
||||
banner.hidden = true;
|
||||
return;
|
||||
}
|
||||
const count = sessions.length;
|
||||
const text = this.$('rebootRestoreBannerText');
|
||||
if (text) {
|
||||
const noun = count === 1 ? 'session' : 'sessions';
|
||||
text.textContent = `Restore ${count} ${noun} from before the reboot`;
|
||||
}
|
||||
const detail = this.$('rebootRestoreBannerDetail');
|
||||
if (detail) {
|
||||
// Names, so the user can tell what they are about to relaunch.
|
||||
const names = sessions
|
||||
.map((s) => s.name || s.workingDir?.split('/').pop() || s.id.slice(0, 8))
|
||||
.slice(0, 4)
|
||||
.join(', ');
|
||||
detail.textContent = count > 4 ? `${names}, …` : names;
|
||||
detail.title = sessions.map((s) => `${s.name || s.id}\n${s.workingDir}`).join('\n\n');
|
||||
}
|
||||
const accept = this.$('rebootRestoreBannerAccept');
|
||||
// The note is hidden at phone width, so the warning travels on the button too.
|
||||
if (accept) accept.title = 'Conversations return; terminal history does not.';
|
||||
banner.hidden = false;
|
||||
},
|
||||
|
||||
/** Rebuild everything on offer. The panes are new, so scrollback does not come back. */
|
||||
async restoreRebootSessions() {
|
||||
const button = this.$('rebootRestoreBannerAccept');
|
||||
if (button) button.disabled = true;
|
||||
const res = await this._api('/api/reboot-restore/restore', { method: 'POST', body: {} });
|
||||
if (res && res.status === 409) {
|
||||
if (button) button.disabled = false;
|
||||
this.showToast?.('A restore is already running', 'info');
|
||||
return;
|
||||
}
|
||||
// The uniform envelope wraps every /api payload; reading the outer object
|
||||
// would report every count as zero.
|
||||
const body = res && res.ok ? (await res.json().catch(() => null))?.data : null;
|
||||
if (!body) {
|
||||
if (button) button.disabled = false;
|
||||
this.showToast?.('Could not restore the sessions', 'error');
|
||||
return;
|
||||
}
|
||||
const restored = body.restored?.length ?? 0;
|
||||
const skipped = body.skipped?.length ?? 0;
|
||||
// Re-read rather than clearing: the server puts back anything it could not
|
||||
// build for a reason that may pass, such as a session limit or an agent that
|
||||
// would not start, and blanking the banner here would put those entries out
|
||||
// of reach until a reload.
|
||||
await this.refreshRebootRestoreBanner();
|
||||
if (button) button.disabled = false;
|
||||
if (restored > 0) {
|
||||
const noun = restored === 1 ? 'conversation' : 'conversations';
|
||||
this.showToast?.(`Restored ${restored} ${noun}. Terminal history did not survive the reboot.`, 'success');
|
||||
}
|
||||
if (skipped > 0) {
|
||||
// Each reason means a different next step for the user, so they are not
|
||||
// collapsed into one message: capacity clears by closing something, a
|
||||
// failed start usually means the CLI is not on the server's PATH.
|
||||
const reasons = new Set((body.skipped ?? []).map((s) => s.reason));
|
||||
this.showToast?.(`${skipped} not restored: ${[...reasons].map(rebootSkipReason).join('; ')}`, 'warning');
|
||||
}
|
||||
},
|
||||
|
||||
/** Re-read the offer after a reconnect, for a tab that was open across the reboot. */
|
||||
async refreshRebootRestoreBanner() {
|
||||
const data = await this._apiJson('/api/reboot-restore');
|
||||
this._rebootRestoreSessions = data?.sessions ?? [];
|
||||
this.renderRebootRestoreBanner();
|
||||
},
|
||||
|
||||
/** Drop the offer. The Resume list still reaches every one of these conversations. */
|
||||
async dismissRebootRestore() {
|
||||
this._rebootRestoreSessions = [];
|
||||
this.renderRebootRestoreBanner();
|
||||
await this._apiPost('/api/reboot-restore/dismiss', {});
|
||||
},
|
||||
});
|
||||
+802
-105
File diff suppressed because it is too large
Load Diff
@@ -395,6 +395,13 @@ Object.assign(CodemanApp.prototype, {
|
||||
document.getElementById('appSettingsShowUltracodeAgents').checked = settings.showUltracodeAgents ?? defaults.showUltracodeAgents ?? false;
|
||||
// Approvals Inbox: synced, default OFF (opt-in; only an explicit true enables).
|
||||
document.getElementById('appSettingsApprovalsInbox').checked = settings.approvalsInboxEnabled === true;
|
||||
// Custom Model Endpoint Profiles: synced, default OFF. The toggle governs both
|
||||
// the Run-menu picker's generated entries and this settings panel's visibility;
|
||||
// the endpoint list itself is server state, loaded on demand below.
|
||||
document.getElementById('appSettingsCustomModelEndpoints').checked = settings.customModelEndpointsEnabled === true;
|
||||
// Assigning .checked above does not fire onchange, so the body's visibility
|
||||
// (and its lazy load) needs an explicit sync on every open, not just a save.
|
||||
this.applyCustomModelEndpointsVisibility();
|
||||
// Read My Mind: synced, default OFF (opt-in; capture + prediction cost real tokens).
|
||||
document.getElementById('appSettingsReadMyMind').checked = settings.readMyMindEnabled === true;
|
||||
document.getElementById('appSettingsUltracodeFloatingWindows').checked =
|
||||
@@ -408,6 +415,8 @@ Object.assign(CodemanApp.prototype, {
|
||||
// header), so the row is hidden elsewhere rather than offering a toggle that
|
||||
// changes nothing. Default ON — only an explicit false turns it off.
|
||||
document.getElementById('appSettingsLineageLines').checked = settings.sessionLineageLines ?? defaults.sessionLineageLines ?? true;
|
||||
// Auto-name sessions: synced, default OFF (opt-in; only an explicit true enables).
|
||||
document.getElementById('appSettingsAutoNameSessions').checked = settings.autoNameSessions === true;
|
||||
const lineageItem = document.getElementById('appSettingsLineageLinesItem');
|
||||
if (lineageItem) lineageItem.style.display = MobileDetection.getDeviceType() === 'desktop' ? '' : 'none';
|
||||
document.getElementById('appSettingsMobileOverview').checked = settings.mobileOverviewEnabled ?? defaults.mobileOverviewEnabled ?? false;
|
||||
@@ -507,6 +516,9 @@ Object.assign(CodemanApp.prototype, {
|
||||
document.getElementById('appSettingsNiceValue').value = niceSettings.niceValue ?? 10;
|
||||
// Model configuration (loaded from server)
|
||||
this.loadModelConfigForSettings();
|
||||
// Custom Model Endpoint Profiles' own load is gated on the toggle above (see
|
||||
// applyCustomModelEndpointsVisibility) — unlike model config, this GET is
|
||||
// pointless work with the feature off, so it is not fired unconditionally.
|
||||
// Notification settings
|
||||
const notifPrefs = this.notificationManager?.preferences || {};
|
||||
document.getElementById('appSettingsNotifEnabled').checked = notifPrefs.enabled ?? true;
|
||||
@@ -2104,6 +2116,7 @@ Object.assign(CodemanApp.prototype, {
|
||||
showSubagents: document.getElementById('appSettingsShowSubagents').checked,
|
||||
showUltracodeAgents: document.getElementById('appSettingsShowUltracodeAgents').checked,
|
||||
approvalsInboxEnabled: document.getElementById('appSettingsApprovalsInbox').checked,
|
||||
customModelEndpointsEnabled: document.getElementById('appSettingsCustomModelEndpoints').checked,
|
||||
readMyMindEnabled: document.getElementById('appSettingsReadMyMind').checked,
|
||||
ultracodeFloatingWindows: document.getElementById('appSettingsUltracodeFloatingWindows').checked,
|
||||
showMultiMonitorButton: document.getElementById('appSettingsShowMultiMonitorButton').checked,
|
||||
@@ -2111,6 +2124,7 @@ Object.assign(CodemanApp.prototype, {
|
||||
showRedrawButton: document.getElementById('appSettingsShowRedrawButton').checked,
|
||||
mobileOverviewEnabled: document.getElementById('appSettingsMobileOverview').checked,
|
||||
sessionLineageLines: document.getElementById('appSettingsLineageLines').checked,
|
||||
autoNameSessions: document.getElementById('appSettingsAutoNameSessions').checked,
|
||||
showSessionButton: document.getElementById('appSettingsShowSessionButton').checked,
|
||||
showAwayDigestButton: document.getElementById('appSettingsShowAwayDigestButton').checked,
|
||||
showCronButton: document.getElementById('appSettingsShowCronButton').checked,
|
||||
@@ -2484,6 +2498,209 @@ Object.assign(CodemanApp.prototype, {
|
||||
}
|
||||
},
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Custom Model Endpoint Profiles (docs/custom-model-endpoints-plan.md)
|
||||
//
|
||||
// CRUD against /api/model-endpoints, rendered into the Models settings section.
|
||||
// Deliberately its own load/save pair rather than folded into openAppSettings/
|
||||
// saveAppSettings: these are server-side infra records (like remote/docker
|
||||
// hosts), not a settings-payload field, so the app-settings-structure guard's
|
||||
// by-id contract does not apply to them — only the `customModelEndpointsEnabled`
|
||||
// toggle itself goes through that path.
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
/**
|
||||
* Toggles the endpoint-management body's visibility to match the setting and,
|
||||
* turning it on, lazily loads the endpoint list. Assigning `.checked` (as the
|
||||
* settings load path does) fires no `change` event, so this must be called
|
||||
* explicitly on open as well as wired to the checkbox's own onchange — a
|
||||
* gate that only worked one of those two ways would show a stale "off"
|
||||
* body right after opening, or a stale "on" one right after saving it off.
|
||||
* With the feature off the body is a list of controls that do nothing, so it
|
||||
* is hidden entirely rather than shown disabled.
|
||||
*/
|
||||
applyCustomModelEndpointsVisibility() {
|
||||
const enabled = document.getElementById('appSettingsCustomModelEndpoints').checked;
|
||||
const body = document.getElementById('customModelEndpointsBody');
|
||||
if (body) body.style.display = enabled ? '' : 'none';
|
||||
if (enabled) this.loadCustomModelEndpointsForSettings();
|
||||
else this.closeCustomModelHostEditor();
|
||||
this._applyCustomModelAdminGate();
|
||||
},
|
||||
|
||||
/**
|
||||
* Endpoint writes are admin-only in multi-user mode (custom-model-routes.ts),
|
||||
* and GET already answers a non-admin with an empty list, which hides every
|
||||
* per-row Edit/Discover/Delete button on its own. The "+ Add endpoint" button
|
||||
* has no row to hide behind, so it needs its own gate — otherwise a non-admin
|
||||
* can open the form, fill it in, and get a 403 toast on Save. Wired to the
|
||||
* `codeman:me` event (admin-ui.js) as well as called from
|
||||
* applyCustomModelEndpointsVisibility(), because `window.__codemanUser`'s
|
||||
* real role can resolve AFTER settings have already been opened once.
|
||||
*/
|
||||
_applyCustomModelAdminGate() {
|
||||
const addBtn = document.getElementById('customModelHostAddBtn');
|
||||
if (!addBtn) return;
|
||||
const me = window.__codemanUser || {};
|
||||
const blocked = me.multiUser && me.role !== 'admin';
|
||||
addBtn.style.display = blocked ? 'none' : '';
|
||||
},
|
||||
|
||||
async loadCustomModelEndpointsForSettings() {
|
||||
// GET /api/model-endpoints wraps its body in the { success, data } envelope
|
||||
// like every other /api route (server.ts's preSerialization hook applies to
|
||||
// arrays too) — _apiJson() unwraps it. A raw fetch().json() here would
|
||||
// silently see the envelope object instead of the array and this panel
|
||||
// would read as "No endpoints yet" forever, even with endpoints saved.
|
||||
const hosts = await this._apiJson('/api/model-endpoints');
|
||||
this._customModelHosts = Array.isArray(hosts) ? hosts : [];
|
||||
this.renderCustomModelHostsList();
|
||||
},
|
||||
|
||||
renderCustomModelHostsList() {
|
||||
const list = document.getElementById('customModelHostsList');
|
||||
if (!list) return;
|
||||
const hosts = this._customModelHosts || [];
|
||||
if (hosts.length === 0) {
|
||||
list.innerHTML = '<p class="set-group-hint">No endpoints yet. Add one below to point a harness at a local or cloud OpenAI-compatible server.</p>';
|
||||
return;
|
||||
}
|
||||
list.innerHTML = hosts
|
||||
.map((h) => {
|
||||
const modelCount = (h.models || []).length;
|
||||
const modelSummary = modelCount === 0
|
||||
? 'No models discovered yet'
|
||||
: `${modelCount} model${modelCount === 1 ? '' : 's'}${h.defaultModelId ? ` · default: ${escapeHtml(h.defaultModelId)}` : ' · no default set'}`;
|
||||
// escapeHtml(JSON.stringify(h.id)) — not JSON.stringify(h.id) alone —
|
||||
// because JSON.stringify's own double quotes would otherwise terminate
|
||||
// this double-quoted attribute at the first one, and everything after
|
||||
// parses as raw tag content rather than the rest of the quoted string.
|
||||
// Same idiom as deleteCase's onclick in session-ui.js. h.id is
|
||||
// regex-constrained server-side (safe either way) but the pattern must
|
||||
// match everywhere it is used, including where the argument is not.
|
||||
const idArg = escapeHtml(JSON.stringify(h.id));
|
||||
return `
|
||||
<div class="set-row" data-endpoint-id="${escapeHtml(h.id)}">
|
||||
<div class="set-row-text">
|
||||
<span class="set-row-label">${escapeHtml(h.label)}</span>
|
||||
<span class="set-row-desc">${escapeHtml(h.baseUrl)} — ${modelSummary}</span>
|
||||
</div>
|
||||
<div class="set-row-actions">
|
||||
<button type="button" class="btn-toolbar btn-sm" onclick="app.discoverCustomModelHostModels(${idArg})">Discover</button>
|
||||
<button type="button" class="btn-toolbar btn-sm" onclick="app.openCustomModelHostEditor(${idArg})">Edit</button>
|
||||
<button type="button" class="btn-toolbar btn-danger btn-sm" onclick="app.deleteCustomModelHost(${idArg})">Delete</button>
|
||||
</div>
|
||||
</div>`;
|
||||
})
|
||||
.join('');
|
||||
},
|
||||
|
||||
/** Opens the inline add/edit form. Pass no id to add a new endpoint. */
|
||||
openCustomModelHostEditor(hostId) {
|
||||
const host = hostId ? (this._customModelHosts || []).find((h) => h.id === hostId) : null;
|
||||
this._editingCustomModelHostId = host ? host.id : null;
|
||||
document.getElementById('customModelHostEditorTitle').textContent = host ? `Edit ${host.label}` : 'Add endpoint';
|
||||
document.getElementById('customModelHostId').value = host?.id || '';
|
||||
document.getElementById('customModelHostId').disabled = !!host; // id is immutable once created
|
||||
document.getElementById('customModelHostLabel').value = host?.label || '';
|
||||
document.getElementById('customModelHostBaseUrl').value = host?.baseUrl || '';
|
||||
document.getElementById('customModelHostApiKey').value = ''; // the server never returns the real value (apiKeySet is a bool)
|
||||
document.getElementById('customModelHostApiKey').placeholder = host?.apiKeySet ? '•••••••• (unchanged if left blank)' : '';
|
||||
document.getElementById('customModelHostAuthStyle').value = host?.authStyle || 'bearer';
|
||||
this._populateCustomModelDefaultSelect(host);
|
||||
document.getElementById('customModelHostEditor').style.display = '';
|
||||
},
|
||||
|
||||
closeCustomModelHostEditor() {
|
||||
document.getElementById('customModelHostEditor').style.display = 'none';
|
||||
this._editingCustomModelHostId = null;
|
||||
},
|
||||
|
||||
_populateCustomModelDefaultSelect(host) {
|
||||
const select = document.getElementById('customModelHostDefaultModel');
|
||||
const models = host?.models || [];
|
||||
select.innerHTML =
|
||||
'<option value="">No default (picker uses the first discovered model)</option>' +
|
||||
models.map((m) => `<option value="${escapeHtml(m)}">${escapeHtml(m)}</option>`).join('');
|
||||
select.value = host?.defaultModelId || '';
|
||||
select.disabled = models.length === 0;
|
||||
},
|
||||
|
||||
async saveCustomModelHostFromEditor() {
|
||||
const id = document.getElementById('customModelHostId').value.trim();
|
||||
const label = document.getElementById('customModelHostLabel').value.trim();
|
||||
const baseUrl = document.getElementById('customModelHostBaseUrl').value.trim();
|
||||
const apiKeyInput = document.getElementById('customModelHostApiKey').value;
|
||||
const authStyle = document.getElementById('customModelHostAuthStyle').value;
|
||||
const defaultModelId = document.getElementById('customModelHostDefaultModel').value || undefined;
|
||||
if (!id || !label || !baseUrl) {
|
||||
this.showToast('Id, label and base URL are all required', 'warning');
|
||||
return;
|
||||
}
|
||||
const editing = this._editingCustomModelHostId;
|
||||
// PUT (server-side) treats an absent apiKey as "keep the stored one" — the
|
||||
// browser never holds the real value to resend deliberately unchanged (see
|
||||
// openCustomModelHostEditor and custom-model-routes.ts's applyStoredApiKey),
|
||||
// so a blank field here means omitting the key entirely, not resending
|
||||
// something we do not have. models/lastDiscoveredAt DO still need
|
||||
// re-sending: PUT replaces the whole record, and this cached copy still
|
||||
// carries both (only apiKey is redacted from what GET hands back).
|
||||
const existing = editing ? (this._customModelHosts || []).find((h) => h.id === editing) : null;
|
||||
const body = {
|
||||
id,
|
||||
label,
|
||||
baseUrl,
|
||||
authStyle,
|
||||
defaultModelId,
|
||||
apiKey: apiKeyInput || undefined,
|
||||
models: existing?.models,
|
||||
lastDiscoveredAt: existing?.lastDiscoveredAt,
|
||||
};
|
||||
try {
|
||||
const res = await fetch(editing ? `/api/model-endpoints/${encodeURIComponent(editing)}` : '/api/model-endpoints', {
|
||||
method: editing ? 'PUT' : 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify(body),
|
||||
});
|
||||
const data = await res.json();
|
||||
if (!data.success) {
|
||||
this.showToast(data.error || 'Failed to save endpoint', 'error');
|
||||
return;
|
||||
}
|
||||
this.showToast(editing ? 'Endpoint updated' : 'Endpoint added', 'success');
|
||||
this.closeCustomModelHostEditor();
|
||||
await this.loadCustomModelEndpointsForSettings();
|
||||
} catch (err) {
|
||||
this.showToast(`Failed to save endpoint: ${err.message}`, 'error');
|
||||
}
|
||||
},
|
||||
|
||||
async discoverCustomModelHostModels(hostId) {
|
||||
this.showToast('Discovering models…', 'info');
|
||||
try {
|
||||
const res = await fetch(`/api/model-endpoints/${encodeURIComponent(hostId)}/discover-models`, { method: 'POST' });
|
||||
const data = await res.json();
|
||||
if (!data.success) {
|
||||
this.showToast(data.error || 'Discovery failed', 'error');
|
||||
return;
|
||||
}
|
||||
this.showToast(`Found ${data.data.models.length} model${data.data.models.length === 1 ? '' : 's'}`, 'success');
|
||||
await this.loadCustomModelEndpointsForSettings();
|
||||
} catch (err) {
|
||||
this.showToast(`Discovery failed: ${err.message}`, 'error');
|
||||
}
|
||||
},
|
||||
|
||||
async deleteCustomModelHost(hostId) {
|
||||
const host = (this._customModelHosts || []).find((h) => h.id === hostId);
|
||||
if (!confirm(`Delete endpoint "${host?.label || hostId}"? Any session currently pointed at it keeps running until cleared.`)) return;
|
||||
try {
|
||||
await fetch(`/api/model-endpoints/${encodeURIComponent(hostId)}`, { method: 'DELETE' });
|
||||
await this.loadCustomModelEndpointsForSettings();
|
||||
} catch (err) {
|
||||
this.showToast(`Failed to delete endpoint: ${err.message}`, 'error');
|
||||
}
|
||||
},
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Visibility Settings & Device-Specific Defaults
|
||||
@@ -3540,3 +3757,15 @@ Object.assign(CodemanApp.prototype, {
|
||||
this.subagentPanelVisible = false;
|
||||
},
|
||||
});
|
||||
|
||||
// window.__codemanUser's real role can resolve after settings have already been
|
||||
// opened once (admin-ui.js fetches /api/me asynchronously and dispatches this on
|
||||
// arrival), so the Custom Model Endpoints admin gate needs to be re-applied when
|
||||
// it does, not just when the modal opens. Optional chaining on addEventListener
|
||||
// itself: several frontend tests (run-mode-ui.test.ts) load this file into a vm
|
||||
// context with a minimal fake `document` that has no event-target methods at
|
||||
// all, and a module-level statement that throws there fails the whole file's
|
||||
// evaluation, not just this feature.
|
||||
document.addEventListener?.('codeman:me', () => {
|
||||
window.app?._applyCustomModelAdminGate?.();
|
||||
});
|
||||
|
||||
@@ -2548,6 +2548,10 @@ body.solo-mode .header-tokens,
|
||||
body.solo-mode .btn-notifications,
|
||||
body.solo-mode .btn-multimonitor,
|
||||
body.solo-mode .header-plan-usage,
|
||||
/* A solo window shows ONE session and has no tab strip to put restored ones in,
|
||||
so offering to rebuild a list of them there is an offer it cannot show the
|
||||
result of. The dashboard that spawned this window carries the banner. */
|
||||
body.solo-mode .reboot-restore-banner,
|
||||
body.solo-mode .btn-lifecycle-log {
|
||||
display: none !important;
|
||||
}
|
||||
@@ -6803,6 +6807,67 @@ body.touch-device .terminal-container .xterm .xterm-helper-textarea {
|
||||
min-height: 0;
|
||||
}
|
||||
|
||||
/* Custom Model Endpoint Profiles' "which model" picker: same bounded-height +
|
||||
scrollable-body shape as .modal-lg above, scoped by id rather than added to
|
||||
.modal-sm itself (three other modals share that class for short, fixed
|
||||
content and do not need a height cap). Without this the modal had no
|
||||
max-height at all, so an endpoint with many discovered models grew the
|
||||
dialog past the viewport with nothing to scroll — "the whole page" and
|
||||
"the list is truncated" turned out to be one and the same bug. `min(70vh,
|
||||
520px)` scales with the monitor (a phone gets 70% of its height, a 4K
|
||||
display never gets a needlessly tall dialog) rather than a fixed value
|
||||
that would be wrong at one end or the other. */
|
||||
#customModelPickModal .modal-content {
|
||||
max-height: min(70vh, 520px);
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
}
|
||||
|
||||
#customModelPickModal .modal-body {
|
||||
overflow-y: auto;
|
||||
flex: 1;
|
||||
min-height: 0;
|
||||
}
|
||||
|
||||
/* Custom Model Endpoint Profiles: llama-swap model-swap confirmation — replaces a native
|
||||
confirm() popup (docs/custom-model-endpoints-plan.md) so it looks and feels like the
|
||||
rest of the app instead of a browser chrome dialog. Shares the context-window-too-small
|
||||
modal's fixes below since both can appear mid-launch, in the same spot, for the same
|
||||
reason — including this rule itself: there is no bare `.modal-footer` base style
|
||||
anywhere in this file, and `.btn-toolbar` is `display: flex` (a block-level flex
|
||||
container with no explicit `inline-flex`), so with no row layout of its own each
|
||||
button took its own full-width line and the two stacked instead of sitting side by
|
||||
side. Centred rather than flex-end per feedback — a two-button Cancel/confirm footer
|
||||
reads better centred than pinned to one edge. */
|
||||
#customModelSwapConfirmModal .modal-footer,
|
||||
#customModelContextWarningModal .modal-footer {
|
||||
display: flex;
|
||||
justify-content: center;
|
||||
gap: 0.5rem;
|
||||
padding: 0.75rem 1rem;
|
||||
border-top: 1px solid var(--border-color);
|
||||
}
|
||||
|
||||
/* Both dialogs can appear while the centred llama-swap status banner (10001, see
|
||||
.center-status-banner) is still on screen — right after "Claude started — switching
|
||||
to llama-swap…" — and .modal's own z-index (1000) sat well under it, so the dialog
|
||||
rendered fully hidden behind the banner (confirmed live, reported against the
|
||||
context-window one but structurally identical for the swap-confirm modal too). */
|
||||
#customModelSwapConfirmModal,
|
||||
#customModelContextWarningModal {
|
||||
z-index: 10010;
|
||||
}
|
||||
|
||||
/* Both messages ARE the modal's whole explanatory content, not a one-line caption under
|
||||
a form field, so .form-hint's 0.65rem caption size (right for what it was designed for)
|
||||
read as illegibly small here, worst on the multi-sentence context-window explanation. */
|
||||
#customModelSwapConfirmMessage,
|
||||
#customModelContextWarningMessage {
|
||||
font-size: 0.85rem;
|
||||
line-height: 1.5;
|
||||
color: var(--text);
|
||||
}
|
||||
|
||||
|
||||
/* Mobile Case Picker - Base Styles */
|
||||
.mobile-case-picker-sheet {
|
||||
@@ -8441,6 +8506,9 @@ kbd {
|
||||
}
|
||||
|
||||
.toast {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: 0.5rem;
|
||||
background: var(--bg-card);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 6px;
|
||||
@@ -8452,6 +8520,7 @@ kbd {
|
||||
opacity: 0;
|
||||
transition: all 0.2s ease;
|
||||
pointer-events: auto;
|
||||
max-width: 420px;
|
||||
}
|
||||
|
||||
.toast.show {
|
||||
@@ -8459,6 +8528,149 @@ kbd {
|
||||
opacity: 1;
|
||||
}
|
||||
|
||||
.toast-message {
|
||||
flex: 1;
|
||||
/* A sticky toast (showToast's opts.duration: 0) can carry a longer, specific
|
||||
message — let it wrap instead of clipping. */
|
||||
white-space: pre-wrap;
|
||||
word-break: break-word;
|
||||
}
|
||||
|
||||
/* Every toast gets one, sticky or not: a sticky toast with no way to close it
|
||||
would just accumulate on screen across repeated failures. */
|
||||
.toast-close {
|
||||
flex-shrink: 0;
|
||||
background: none;
|
||||
border: none;
|
||||
color: inherit;
|
||||
opacity: 0.6;
|
||||
font-size: 1.1rem;
|
||||
line-height: 1;
|
||||
padding: 0 0.15rem;
|
||||
cursor: pointer;
|
||||
}
|
||||
|
||||
.toast-close:hover {
|
||||
opacity: 1;
|
||||
}
|
||||
|
||||
/* Custom Model Endpoint Profiles: the "switching backends" / "loading model" states
|
||||
(docs/custom-model-endpoints-plan.md) — a small set of messages prominent and
|
||||
screen-centred rather than corner toasts, since they can sit on screen for well
|
||||
over a minute (a real llama-swap model load) and are easy to mistake for nothing
|
||||
happening. Non-blocking: `pointer-events: none` on the wrapper (no backdrop, no
|
||||
click-catcher) with `auto` restored only on the card itself, purely so the text
|
||||
inside remains selectable — there is nothing to click to dismiss it early. */
|
||||
.center-status-banner {
|
||||
position: fixed;
|
||||
top: 50%;
|
||||
left: 50%;
|
||||
transform: translate(-50%, -50%) scale(0.96);
|
||||
z-index: 10001;
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: 0.75rem;
|
||||
background: var(--bg-card);
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 10px;
|
||||
padding: 1rem 1.5rem;
|
||||
box-shadow: 0 8px 32px rgba(0, 0, 0, 0.4);
|
||||
font-size: 0.95rem;
|
||||
font-weight: 500;
|
||||
color: var(--text);
|
||||
max-width: min(90vw, 460px);
|
||||
text-align: left;
|
||||
opacity: 0;
|
||||
pointer-events: none;
|
||||
transition:
|
||||
opacity 0.2s ease,
|
||||
transform 0.2s ease;
|
||||
}
|
||||
|
||||
.center-status-banner.show {
|
||||
opacity: 1;
|
||||
transform: translate(-50%, -50%) scale(1);
|
||||
}
|
||||
|
||||
/* `hidden` has to be re-asserted over the `display: flex` above, or `dismiss()`
|
||||
setting `el.hidden = true` does nothing (same trap as `.home-sessions[hidden]`
|
||||
below): the card stays laid out at `opacity: 0` with its text/cancel/close
|
||||
children still `pointer-events: auto`, an invisible click-blocker dead centre
|
||||
over the terminal until the page reloads. */
|
||||
.center-status-banner[hidden] {
|
||||
display: none;
|
||||
}
|
||||
|
||||
.center-status-spinner {
|
||||
flex-shrink: 0;
|
||||
width: 18px;
|
||||
height: 18px;
|
||||
border-radius: 50%;
|
||||
border: 2px solid var(--border);
|
||||
border-top-color: var(--accent, var(--text));
|
||||
animation: center-status-spin 0.8s linear infinite;
|
||||
}
|
||||
|
||||
@keyframes center-status-spin {
|
||||
to {
|
||||
transform: rotate(360deg);
|
||||
}
|
||||
}
|
||||
|
||||
.center-status-text {
|
||||
flex: 1;
|
||||
pointer-events: auto;
|
||||
white-space: pre-wrap;
|
||||
word-break: break-word;
|
||||
}
|
||||
|
||||
/* Error variant: the load didn't finish in time — nothing is "in progress" anymore (no
|
||||
spinner), and since this one doesn't dismiss itself, it needs a close button the user
|
||||
can actually click, so pointer-events is restored here too (see the wrapper's own
|
||||
comment on why that's `none` by default). */
|
||||
.center-status-error {
|
||||
border-color: rgba(239, 68, 68, 0.5);
|
||||
}
|
||||
|
||||
.center-status-close {
|
||||
flex-shrink: 0;
|
||||
pointer-events: auto;
|
||||
background: none;
|
||||
border: none;
|
||||
color: inherit;
|
||||
opacity: 0.6;
|
||||
font-size: 1.2rem;
|
||||
line-height: 1;
|
||||
padding: 0 0.15rem;
|
||||
cursor: pointer;
|
||||
}
|
||||
|
||||
.center-status-close:hover {
|
||||
opacity: 1;
|
||||
}
|
||||
|
||||
/* The Cancel button on an 'info' banner (e.g. the model-loading banner) — a real button
|
||||
rather than the bare "×" close glyph above, since "Cancel" is an action with a
|
||||
consequence (the caller's onCancel closes a session), not a plain dismiss. */
|
||||
.center-status-cancel {
|
||||
flex-shrink: 0;
|
||||
pointer-events: auto;
|
||||
background: none;
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 6px;
|
||||
color: inherit;
|
||||
opacity: 0.75;
|
||||
font-size: 0.8rem;
|
||||
font-weight: 500;
|
||||
padding: 0.25rem 0.6rem;
|
||||
cursor: pointer;
|
||||
}
|
||||
|
||||
.center-status-cancel:hover {
|
||||
opacity: 1;
|
||||
border-color: var(--text-muted, var(--border));
|
||||
}
|
||||
|
||||
.toast-success { border-color: rgba(34, 197, 94, 0.4); }
|
||||
.toast-error { border-color: rgba(239, 68, 68, 0.4); }
|
||||
.toast-warning { border-color: rgba(234, 179, 8, 0.4); }
|
||||
@@ -15134,6 +15346,12 @@ html[data-skin="daylight-blue"] .welcome-btn-tunnel.active:hover {
|
||||
|
||||
.run-mode-dot.web { background: #38bdf8; }
|
||||
.run-mode-webviews { max-height: 180px; overflow-y: auto; }
|
||||
/* Custom Model Endpoint Profiles' generated entries: `.run-mode-menu.active`'s
|
||||
own `gap: 2px` only spaces its DIRECT children, and this container (like
|
||||
`.run-mode-webviews` above) is one such child holding several buttons of
|
||||
its own, so it needs the same gap repeated one level down or its rows sit
|
||||
flush against each other. */
|
||||
.run-mode-custom-models { display: flex; flex-direction: column; gap: 2px; }
|
||||
|
||||
/* A saved URL is a ROW: open on the left, edit + delete on the right, so a URL can
|
||||
be changed or removed without first opening it as a tab. The side buttons stay
|
||||
@@ -15243,6 +15461,85 @@ html[data-skin="daylight-blue"] .welcome-btn-tunnel.active:hover {
|
||||
skin, including the light ones. Visibility is driven by the `hidden`
|
||||
attribute, so the display rules need !important to lose to it. */
|
||||
|
||||
/* Reboot-restore offer. Amber rather than red: nothing is wrong, the board is
|
||||
asking a question, and the user can ignore it. See reboot-restore-ui.js. */
|
||||
.reboot-restore-banner {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
gap: 0.6rem;
|
||||
padding: 0.45rem 1rem;
|
||||
background: linear-gradient(90deg, #b45309, #92400e);
|
||||
border-bottom: 1px solid rgba(0, 0, 0, 0.35);
|
||||
color: #fff;
|
||||
font-size: 0.78rem;
|
||||
font-weight: 600;
|
||||
letter-spacing: 0.01em;
|
||||
flex-shrink: 0;
|
||||
z-index: 1250;
|
||||
}
|
||||
|
||||
.reboot-restore-banner[hidden] {
|
||||
display: none !important;
|
||||
}
|
||||
|
||||
.reboot-restore-banner-icon {
|
||||
flex-shrink: 0;
|
||||
font-size: 0.95rem;
|
||||
line-height: 1;
|
||||
}
|
||||
|
||||
.reboot-restore-banner-text {
|
||||
white-space: nowrap;
|
||||
}
|
||||
|
||||
.reboot-restore-banner-detail {
|
||||
color: rgba(255, 255, 255, 0.8);
|
||||
font-weight: 500;
|
||||
/* A flex item will not shrink below its content width at the default
|
||||
`min-width: auto`, so without this the session names push the buttons out of
|
||||
the line between the phone breakpoint and full width. */
|
||||
min-width: 0;
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
}
|
||||
|
||||
.reboot-restore-banner-note {
|
||||
color: rgba(255, 255, 255, 0.75);
|
||||
font-weight: 500;
|
||||
white-space: nowrap;
|
||||
margin-left: auto;
|
||||
}
|
||||
|
||||
.reboot-restore-banner-accept,
|
||||
.reboot-restore-banner-dismiss {
|
||||
flex-shrink: 0;
|
||||
padding: 0.2rem 0.6rem;
|
||||
border-radius: 5px;
|
||||
border: 1px solid rgba(255, 255, 255, 0.55);
|
||||
background: rgba(255, 255, 255, 0.12);
|
||||
color: #fff;
|
||||
font-size: 0.72rem;
|
||||
font-weight: 600;
|
||||
cursor: pointer;
|
||||
}
|
||||
|
||||
.reboot-restore-banner-accept:hover,
|
||||
.reboot-restore-banner-dismiss:hover {
|
||||
background: rgba(255, 255, 255, 0.24);
|
||||
}
|
||||
|
||||
.reboot-restore-banner-accept:disabled {
|
||||
opacity: 0.6;
|
||||
cursor: default;
|
||||
}
|
||||
|
||||
.reboot-restore-banner-dismiss {
|
||||
border-color: rgba(255, 255, 255, 0.3);
|
||||
background: transparent;
|
||||
font-weight: 500;
|
||||
}
|
||||
|
||||
.offline-banner {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
@@ -15306,6 +15603,18 @@ html[data-skin="daylight-blue"] .welcome-btn-tunnel.active:hover {
|
||||
background: rgba(255, 255, 255, 0.24);
|
||||
}
|
||||
|
||||
/* Remote-host unreachable (host asleep, can be woken). Reuses the offline-banner
|
||||
layout and children; amber instead of red because the Codeman session itself is
|
||||
perfectly healthy — only the machine is asleep. */
|
||||
.host-wake-banner {
|
||||
background: linear-gradient(90deg, #b45309, #92400e);
|
||||
}
|
||||
|
||||
.host-wake-banner .offline-banner-retry:disabled {
|
||||
opacity: 0.6;
|
||||
cursor: default;
|
||||
}
|
||||
|
||||
/* Above the mobile fixed header (1200) and modals (1300): this is a blocking
|
||||
"nothing works right now" state, and it only appears before any session
|
||||
state has loaded, so there is no modal underneath to bury. Stays below the
|
||||
@@ -16207,6 +16516,29 @@ html[data-tab-orientation='vertical'] .home-sessions {
|
||||
gap: 3px;
|
||||
}
|
||||
|
||||
/* Custom Model Endpoint Profiles' inline add/edit form: a nested panel rather
|
||||
than a modal, so it needs its own border to read as a distinct sub-section
|
||||
inside .set-group-body's flat row stack. `--control-bg` rather than a
|
||||
hardcoded black alpha — CLAUDE.md records that literal fill turning the
|
||||
settings live preview into a grey slab on the light skins, and this panel
|
||||
sits in the very same modal. */
|
||||
:is(#appSettingsModal, #sessionOptionsModal, #createCaseModal) .set-inline-form {
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 3px;
|
||||
margin-top: 6px;
|
||||
padding: 10px 12px;
|
||||
border: 1px solid var(--border);
|
||||
border-radius: 8px;
|
||||
background: var(--control-bg);
|
||||
}
|
||||
|
||||
:is(#appSettingsModal, #sessionOptionsModal, #createCaseModal) .set-inline-form h5 {
|
||||
margin: 0 0 4px;
|
||||
font-size: 0.72rem;
|
||||
color: var(--text-muted);
|
||||
}
|
||||
|
||||
/* ── rows ─────────────────────────────────────────────────────────────── */
|
||||
:is(#appSettingsModal, #sessionOptionsModal, #createCaseModal) .set-row {
|
||||
display: flex;
|
||||
|
||||
@@ -66,6 +66,38 @@
|
||||
let composing = false;
|
||||
const pending = [];
|
||||
|
||||
/**
|
||||
* Resolve every candidate still pending, right now, instead of waiting for
|
||||
* its zero-delay timer.
|
||||
*
|
||||
* Android soft keyboards commit the last character and send the Enter key
|
||||
* in ONE InputConnection transaction: the `input` event and the Enter
|
||||
* keydown are both processed before any timer runs. Left on its timer the
|
||||
* candidate lost BOTH ways — xterm emits '\r' synchronously from the Enter
|
||||
* keydown (so the local-echo composer submitted the prompt without the
|
||||
* character), and that '\r' bumps `canonicalCount`, so the candidate then
|
||||
* read "xterm spoke for this keystroke" and stood down, dropping the
|
||||
* character outright. That is the "every message loses its last character"
|
||||
* report from phones.
|
||||
*
|
||||
* Draining at the next keydown is correct on both counts: the counter still
|
||||
* holds the value it had while this candidate's keystroke was current, and
|
||||
* the byte reaches the composer ahead of whatever the new key emits.
|
||||
*/
|
||||
function flushPending() {
|
||||
for (const candidate of pending.splice(0)) {
|
||||
if (candidate.timer !== null) {
|
||||
try {
|
||||
clearTimer(candidate.timer);
|
||||
} catch {
|
||||
// A broken timer host must not break input handling.
|
||||
}
|
||||
candidate.timer = null;
|
||||
}
|
||||
resolveCandidate(candidate);
|
||||
}
|
||||
}
|
||||
|
||||
function cancelPending() {
|
||||
for (const candidate of pending.splice(0)) {
|
||||
candidate.active = false;
|
||||
@@ -111,6 +143,11 @@
|
||||
*/
|
||||
function handleKeyEvent(event) {
|
||||
if (destroyed || event?.type !== 'keydown') return;
|
||||
// Settle the PREVIOUS keystroke before this one can move the counter or
|
||||
// reach the PTY — see flushPending(). This runs from xterm's custom key
|
||||
// handler, i.e. before xterm processes the key, so a recovered character
|
||||
// is always ordered ahead of the bytes this keydown produces.
|
||||
flushPending();
|
||||
keydownSnapshot = canonicalCount;
|
||||
}
|
||||
|
||||
|
||||
+162
-27
@@ -364,12 +364,27 @@ Object.assign(CodemanApp.prototype, {
|
||||
// this handler before its own cancel()), so preventDefault is explicit:
|
||||
// without it the browser runs its native copy on top of ours.
|
||||
if (this.shouldCopyTerminalSelectionFromShortcut?.(ev)) {
|
||||
const selection = this.terminal.hasSelection?.() ? this.terminal.getSelection() : '';
|
||||
if (selection) {
|
||||
// The CLEANED selection decides, not the raw one. A drag across the blank
|
||||
// part of a row selects real padding spaces, which are truthy, so testing
|
||||
// the raw text would spend this press on a copy of nothing and make the
|
||||
// user press again to interrupt.
|
||||
const selection = this.cleanedTerminalSelection();
|
||||
if (selection.trim()) {
|
||||
ev.preventDefault();
|
||||
void this.copyTerminalSelection(selection);
|
||||
return false;
|
||||
}
|
||||
// Nothing worth copying. The clear is for feedback, not for the
|
||||
// interrupt: the gate above tests the CLEANED selection, so a
|
||||
// padding-only selection left set cleans to '' on every later press and
|
||||
// falls through to the PTY anyway. What it buys is that a highlight
|
||||
// which copies nothing does not linger with no explanation, which is
|
||||
// also what the toast is for. Falls through exactly as an empty
|
||||
// selection does, so this press still reaches the PTY as 0x03.
|
||||
if (this.terminal?.hasSelection?.()) {
|
||||
this.terminal.clearSelection?.();
|
||||
this.showToast('Nothing to copy', 'warning');
|
||||
}
|
||||
if (ev.shiftKey) {
|
||||
ev.preventDefault();
|
||||
return false;
|
||||
@@ -3159,6 +3174,38 @@ Object.assign(CodemanApp.prototype, {
|
||||
return buffer.viewportY >= buffer.baseY - 2;
|
||||
},
|
||||
|
||||
/**
|
||||
* Re-take the sticky-scroll baseline from where the viewport now sits.
|
||||
*
|
||||
* `batchTerminalWrite` samples `_wasAtBottomBeforeWrite` before it queues
|
||||
* data, and `flushPendingWrites` scrolls to the bottom off that sample. A
|
||||
* buffer load that replays its queue samples at the worst possible moment:
|
||||
* `_finishBufferLoad` runs inside `chunkedTerminalWrite`, before its promise
|
||||
* resolves, with the terminal freshly reset and rewritten, so the sample is
|
||||
* always true. A caller that then restores the reader's position would have
|
||||
* that restore undone by the next flush.
|
||||
*
|
||||
* `_onSessionNeedsRefresh` and `_maybeRefetchFullHistory` restore a position
|
||||
* and both call this, so their baseline describes the position they chose.
|
||||
*
|
||||
* The other two load paths do not call it, for different reasons.
|
||||
* `_onSessionClearTerminal` resets and rewrites with no scroll afterwards,
|
||||
* so the sampled true is already the truth there. `selectSession` does NOT
|
||||
* end at the bottom, whatever its `scrollToBottom()` after the write
|
||||
* suggests: it ends at `scrollToLastNonEmptyLine()`, which targets
|
||||
* `lastNonEmptyLine - rows + 2` and therefore parks ABOVE `baseY` whenever
|
||||
* the replayed frame keeps trailing blank rows, which a full capture does on
|
||||
* purpose. Its baseline is a stale true. What decides whether that matters
|
||||
* is the sticky snap in `flushPendingWrites`, and since de864e7d that snap
|
||||
* fires only when the flush found the viewport already at the bottom
|
||||
* (`preserveViewportY === null`), which a parked selectSession viewport is
|
||||
* not. Do not read the absent call here as a claim that selectSession lands
|
||||
* at the bottom.
|
||||
*/
|
||||
_syncStickyScrollBaseline() {
|
||||
this._wasAtBottomBeforeWrite = this.isTerminalAtBottom();
|
||||
},
|
||||
|
||||
// Record manual scroll gestures so sticky-scroll can give an upward scroll a
|
||||
// short grace window (see _hasRecentUserScrollUp). A downward scroll that
|
||||
// lands back at the bottom clears the suppression immediately.
|
||||
@@ -3295,7 +3342,11 @@ Object.assign(CodemanApp.prototype, {
|
||||
// to prevent interleaving historical buffer data with live SSE data.
|
||||
// This is critical: interleaving causes cursor position chaos with Ink redraws.
|
||||
if (this._isLoadingBuffer) {
|
||||
if (this._loadBufferQueue) this._loadBufferQueue.push(data);
|
||||
// Each entry records when it arrived. A flush of a tmux-capture load
|
||||
// replays only what arrived after the capture; without the timestamp it
|
||||
// would have to replay the whole queue, duplicating the events the
|
||||
// capture already contains. See _finishBufferLoad's `since`.
|
||||
if (this._loadBufferQueue) this._loadBufferQueue.push({ at: performance.now(), data });
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -3809,9 +3860,14 @@ Object.assign(CodemanApp.prototype, {
|
||||
* and a tick-Worker so progress continues on occluded / idle-throttled tabs.
|
||||
* @param {string} buffer - The full terminal buffer to write
|
||||
* @param {number} chunkSize - Size of each chunk (default 32KB)
|
||||
* @param {string} [loadOwner] - Load token to finish under
|
||||
* @param {{ flushQueued?: boolean, since?: number }} [finishOpts] - Passed to
|
||||
* `_finishBufferLoad`. This method ends the load for every non-empty buffer,
|
||||
* so a caller that wants the queue replayed has to say so HERE; the call in
|
||||
* `selectSession` only runs when the write was skipped entirely.
|
||||
* @returns {Promise<{parsedAt: number, bufferLength: number, completed: boolean}>} Parse marker snapshot
|
||||
*/
|
||||
chunkedTerminalWrite(buffer, chunkSize = TERMINAL_CHUNK_SIZE, loadOwner) {
|
||||
chunkedTerminalWrite(buffer, chunkSize = TERMINAL_CHUNK_SIZE, loadOwner, finishOpts) {
|
||||
// Generation counter: if a newer chunkedTerminalWrite starts (tab switch),
|
||||
// older writes abort instead of continuing to push stale data into the terminal.
|
||||
const writeGen = ++this._chunkedWriteGen;
|
||||
@@ -3824,7 +3880,7 @@ Object.assign(CodemanApp.prototype, {
|
||||
completed,
|
||||
});
|
||||
if (!buffer || buffer.length === 0) {
|
||||
this._finishBufferLoad(bufferLoadOwner);
|
||||
this._finishBufferLoad(bufferLoadOwner, finishOpts);
|
||||
resolve(parseSnapshot());
|
||||
return;
|
||||
}
|
||||
@@ -3838,7 +3894,7 @@ Object.assign(CodemanApp.prototype, {
|
||||
this.terminal.write(cleanBuffer, () => resolve(parseSnapshot()));
|
||||
// The write is now ordered in xterm's queue. Release live output before
|
||||
// parsing completes; subsequent writes stay behind it without being lost.
|
||||
this._finishBufferLoad(bufferLoadOwner);
|
||||
this._finishBufferLoad(bufferLoadOwner, finishOpts);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -3869,7 +3925,7 @@ Object.assign(CodemanApp.prototype, {
|
||||
);
|
||||
resolve(result);
|
||||
});
|
||||
this._finishBufferLoad(bufferLoadOwner);
|
||||
this._finishBufferLoad(bufferLoadOwner, finishOpts);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -3883,15 +3939,49 @@ Object.assign(CodemanApp.prototype, {
|
||||
});
|
||||
},
|
||||
|
||||
/**
|
||||
* Open a buffer load: live terminal events are queued from here until
|
||||
* `_finishBufferLoad` decides what to do with them. Returns the load token the
|
||||
* finish call must present; a stale token makes that call a no-op.
|
||||
*
|
||||
* @param {string} [owner] Reuse an existing token to re-enter the same load
|
||||
* (see below); omit it to start a new one.
|
||||
* @returns {string} The load token.
|
||||
*/
|
||||
_beginBufferLoad(owner) {
|
||||
if (this._bufferLoadSeq === undefined) this._bufferLoadSeq = 0;
|
||||
const loadOwner = owner === undefined ? `buffer-${++this._bufferLoadSeq}` : owner;
|
||||
// `selectSession` opens the load before its fetch, and `chunkedTerminalWrite`
|
||||
// opens it again under the SAME owner when it starts writing. Resetting the
|
||||
// queue on that second call would throw away everything that arrived during
|
||||
// the fetch, which on the capture path is output no buffer holds. Re-entering
|
||||
// one load keeps its queue; a genuinely new load still starts empty.
|
||||
const reentering = this._bufferLoadOwner === loadOwner && Array.isArray(this._loadBufferQueue);
|
||||
this._bufferLoadOwner = loadOwner;
|
||||
this._isLoadingBuffer = true;
|
||||
if (!reentering) this._loadBufferQueue = [];
|
||||
return loadOwner;
|
||||
},
|
||||
|
||||
/**
|
||||
* Complete a buffer load: unblock live SSE writes.
|
||||
* Called when chunkedTerminalWrite finishes (or is skipped for empty buffers).
|
||||
*
|
||||
* By default queued SSE events are DISCARDED, not flushed. For an established
|
||||
* session the loaded buffer from the API is the source of truth up to the
|
||||
* response timestamp; SSE events queued during the fetch+write overlap already
|
||||
* appear in that buffer, so flushing them writes duplicate data (especially Ink
|
||||
* cursor-up redraws), corrupting the terminal display.
|
||||
* session whose buffer came from the server's accumulated byte history, that
|
||||
* history is the source of truth up to the response timestamp; SSE events
|
||||
* queued during the fetch+write overlap already appear in it, so flushing
|
||||
* them writes duplicate data (especially Ink cursor-up redraws), corrupting
|
||||
* the terminal display.
|
||||
*
|
||||
* A tmux PANE CAPTURE is the exception, and the reason `since` exists. A
|
||||
* capture is a point-in-time frame taken part-way through the fetch, so it is
|
||||
* the source of truth only up to CAPTURE time — not up to the response. Every
|
||||
* event that arrives between the capture and the end of the chunked write is
|
||||
* queued and, under a plain discard, lost outright: nothing re-fetches, and
|
||||
* the CLI's next partial redraw lands on a frame the terminal never received.
|
||||
* The caller passes the response's own arrival time as `since` so exactly
|
||||
* that tail is replayed and the pre-capture events stay dropped.
|
||||
*
|
||||
* COD-144: a brand-new session is the exception. Its terminal fetch can resolve
|
||||
* BEFORE the PTY emits its first prompt, so the fetched buffer is empty and the
|
||||
@@ -3905,17 +3995,10 @@ Object.assign(CodemanApp.prototype, {
|
||||
* After unblocking, new SSE/WS events deliver subsequent output normally.
|
||||
*
|
||||
* @param {string} [owner] Load token from `_beginBufferLoad`; a stale owner is a no-op.
|
||||
* @param {{ flushQueued?: boolean }} [opts] When `flushQueued` is true, replay any queued events.
|
||||
* @param {{ flushQueued?: boolean, since?: number }} [opts] When `flushQueued`
|
||||
* is true, replay queued events whose arrival timestamp is at or after
|
||||
* `since` (default 0, meaning the whole queue).
|
||||
*/
|
||||
_beginBufferLoad(owner) {
|
||||
if (this._bufferLoadSeq === undefined) this._bufferLoadSeq = 0;
|
||||
const loadOwner = owner === undefined ? `buffer-${++this._bufferLoadSeq}` : owner;
|
||||
this._bufferLoadOwner = loadOwner;
|
||||
this._isLoadingBuffer = true;
|
||||
this._loadBufferQueue = [];
|
||||
return loadOwner;
|
||||
},
|
||||
|
||||
_finishBufferLoad(owner, opts) {
|
||||
if (owner !== undefined && this._bufferLoadOwner !== owner) {
|
||||
return false;
|
||||
@@ -3926,9 +4009,13 @@ Object.assign(CodemanApp.prototype, {
|
||||
this._bufferLoadOwner = null;
|
||||
// COD-144: replay (rather than discard) queued live events when the load
|
||||
// painted nothing — the queued prompt is the only content a new session has.
|
||||
// A tmux-capture load replays too, but only the tail: `since` cuts the queue
|
||||
// at the moment the capture stopped being able to contain what arrived.
|
||||
if (opts?.flushQueued && queued && queued.length) {
|
||||
for (const data of queued) {
|
||||
this.batchTerminalWrite(data);
|
||||
const since = typeof opts.since === 'number' ? opts.since : 0;
|
||||
for (const entry of queued) {
|
||||
if (entry.at < since) continue;
|
||||
this.batchTerminalWrite(entry.data);
|
||||
}
|
||||
}
|
||||
return true;
|
||||
@@ -4069,12 +4156,54 @@ Object.assign(CodemanApp.prototype, {
|
||||
return !ev.altKey && (ev.key || '').toLowerCase() === 'c';
|
||||
},
|
||||
|
||||
/**
|
||||
* xterm's current selection, cleaned for the clipboard. The transform itself
|
||||
* is CodemanCopySelection.clean in constants.js, beside decideAutoCopy; this
|
||||
* is the half that needs the live terminal.
|
||||
*
|
||||
* `text` is for the callers that already read the selection to decide whether
|
||||
* to copy at all (the Ctrl+C gate and the right-click handler), so the read is
|
||||
* not repeated. The transform is idempotent on xterm output, so an
|
||||
* already-cleaned string is an acceptable argument: a CR is consumed by the
|
||||
* parser as a cursor move and never stored in a cell, so the only \r the
|
||||
* selection can carry is the Windows line join, and that is what makes the
|
||||
* trailing scan a fixed point. Fuzzed over 300 000 realistic selections.
|
||||
*
|
||||
* A COLUMN selection comes back untouched. Alt+drag makes one (xterm's
|
||||
* shouldColumnSelect keys on altKey alone, and Codeman sets neither of the
|
||||
* terminals it creates with the one option that would disable it), and a
|
||||
* rectangle's whole point is that its rows line up, which trimming each row
|
||||
* to its own last glyph would destroy. xterm exposes the mode nowhere public,
|
||||
* so this reads the private field the way this file already reads
|
||||
* terminal._core for cell dimensions, and falls back to cleaning normally if
|
||||
* a future xterm renames it. SelectionMode.COLUMN is 3.
|
||||
*/
|
||||
cleanedTerminalSelection(text) {
|
||||
const raw = text ?? (this.terminal?.hasSelection?.() ? this.terminal.getSelection() : '');
|
||||
if (!raw) return '';
|
||||
if (this.terminal?._core?._selectionService?._activeSelectionMode === 3) return raw;
|
||||
const clean = window.CodemanCopySelection?.clean;
|
||||
if (!clean) return raw;
|
||||
return clean(raw);
|
||||
},
|
||||
|
||||
// Copy the current terminal selection. Goes through _copyText (Clipboard API,
|
||||
// then a hidden-textarea + execCommand fallback) because install.sh's LAN
|
||||
// option serves plain HTTP, where navigator.clipboard is undefined.
|
||||
async copyTerminalSelection(text) {
|
||||
const selection = text ?? (this.terminal.hasSelection?.() ? this.terminal.getSelection() : '');
|
||||
if (!selection) return false;
|
||||
const selection = this.cleanedTerminalSelection(text);
|
||||
// trim(), not emptiness: a multi-row drag across padding cleans to newlines
|
||||
// alone, which are truthy, and a bare newline pasted into a chat composer
|
||||
// or a shell submits the line. decideAutoCopy applies the same rule.
|
||||
if (!selection.trim()) {
|
||||
// Clearing is feedback, not protection. The Ctrl+C gate tests the CLEANED
|
||||
// selection, so a padding-only selection left set can no longer swallow a
|
||||
// later interrupt; it cleans to '' and the press reaches the PTY. What the
|
||||
// clear avoids is a highlight that sits there having copied nothing.
|
||||
this.terminal?.clearSelection?.();
|
||||
this.showToast('Nothing to copy', 'warning');
|
||||
return false;
|
||||
}
|
||||
const ok = await this._copyText(selection);
|
||||
if (ok) {
|
||||
// Clearing is what makes a second Ctrl+C an interrupt (and xterm already
|
||||
@@ -4123,9 +4252,15 @@ Object.assign(CodemanApp.prototype, {
|
||||
async _flushAutoCopySelection() {
|
||||
const decide = window.CodemanAutoCopy?.decide;
|
||||
if (!decide || !this.terminal) return;
|
||||
const text = this.terminal.hasSelection?.() ? this.terminal.getSelection() : '';
|
||||
// The toggle is read FIRST because Auto Copy is off by default: reading and
|
||||
// cleaning a selection that can run to the 50 000-row scrollback ceiling
|
||||
// costs real time on a phone, and every mouseup would pay it for nothing.
|
||||
// Cleaning before decide() then means its dedupe and size cap both measure
|
||||
// the text that actually reaches the clipboard, not the padded rows behind.
|
||||
const enabled = this._autoCopySelectionEnabled();
|
||||
const text = enabled ? this.cleanedTerminalSelection() : '';
|
||||
const verdict = decide({
|
||||
enabled: this._autoCopySelectionEnabled(),
|
||||
enabled,
|
||||
text,
|
||||
lastCopied: this._autoCopyLastText,
|
||||
pending: !!this._autoCopyPending,
|
||||
|
||||
@@ -0,0 +1,198 @@
|
||||
/**
|
||||
* @fileoverview The pending restore plan: what a host reboot destroyed, waiting on a click.
|
||||
*
|
||||
* The boot pass builds this plan inside `restoreMuxSessions()`, in the window
|
||||
* where reconciliation has reported the dead sessions and `cleanupStaleSessions()`
|
||||
* has not pruned their records yet. The board then offers "restore N sessions
|
||||
* from before the reboot", and `web/routes/reboot-restore-routes` spends the plan
|
||||
* when the user clicks.
|
||||
*
|
||||
* Invariants:
|
||||
* - Entries are in-memory only. A server restart drops the plan, and nothing
|
||||
* re-builds it, because the records it was built from are pruned by then.
|
||||
* That costs the convenience this feature adds and never the conversation:
|
||||
* the conversation IS the transcript under `~/.claude/projects`, which
|
||||
* `services/unified-session-service.ts` reads for the Welcome screen's Resume
|
||||
* list and the Session Manager, and `resumeHistorySession()` in
|
||||
* `web/public/terminal-ui.js` resumes from a row there with no persisted
|
||||
* session record involved. A dropped plan therefore returns the user to
|
||||
* resuming by hand, one at a time, which is where they are without this
|
||||
* feature. What the plan held that a transcript does not is the owner, the
|
||||
* name, the env overrides, the effort and the lineage.
|
||||
* - Module-level singleton in the style of `web/approval-inbox.ts`: no `Session`
|
||||
* import and no IO, which keeps it unit-testable and cycle-free.
|
||||
* - Spending is take-then-build: `take()` removes entries synchronously, before
|
||||
* the route's first `await`, so a double-click or two devices cannot both
|
||||
* reach the same entry and put two panes on one conversation.
|
||||
* - One restore runs at a time per owner. `beginSpending()` single-flights the
|
||||
* route, so two concurrent clicks cannot interleave pane creation for the same
|
||||
* user, while two different users never block each other.
|
||||
*
|
||||
* @dependencies reboot-restore (RebootRestoreEntry)
|
||||
* @consumedby web/server (plan build at boot), web/routes/reboot-restore-routes
|
||||
*
|
||||
* @module web/reboot-restore-registry
|
||||
*/
|
||||
|
||||
import type { RebootRestoreEntry } from '../reboot-restore.js';
|
||||
|
||||
/**
|
||||
* A plan older than this is dropped on read. A machine that rebooted yesterday
|
||||
* has moved on, and an offer nobody took by then is noise rather than a rescue.
|
||||
*/
|
||||
const PLAN_TTL_MS = 24 * 60 * 60 * 1000;
|
||||
|
||||
export class RebootRestoreRegistry {
|
||||
/** Keyed by session id, in the order the boot pass found them. */
|
||||
private entries = new Map<string, RebootRestoreEntry>();
|
||||
/** When the boot pass built the plan, in ms since the epoch. */
|
||||
private builtAt = 0;
|
||||
/**
|
||||
* Entries handed to a restore that has not finished, by session id, each
|
||||
* remembering which caller is spending it.
|
||||
*
|
||||
* A taken entry is still part of the offer until its restore resolves it, so
|
||||
* it has to stay reachable by everything that can invalidate an offer. Holding
|
||||
* the entries themselves — rather than a counter to compare against later —
|
||||
* means `clear()` filters them by the SAME `canAccess(entry.owner)` predicate
|
||||
* it already applies to the plan. A counter cannot do that, because the caller
|
||||
* spending an entry need not be its owner: an admin may restore another user's
|
||||
* sessions, and then the spender and the owner are different keys.
|
||||
*/
|
||||
private parked = new Map<string, { entry: RebootRestoreEntry; spender: string | undefined }>();
|
||||
/**
|
||||
* Owners with a restore in flight, between its take and its last pane.
|
||||
* Keyed by owner so one user's restore does not turn another user's click into
|
||||
* a conflict; `take()` already guarantees no two callers get the same entry.
|
||||
* Single-user mode has one key, `undefined`, so it behaves as one global flight.
|
||||
*/
|
||||
private spending = new Set<string | undefined>();
|
||||
|
||||
/** Replace the plan with what the boot pass found. An empty list clears it. */
|
||||
set(entries: readonly RebootRestoreEntry[]): void {
|
||||
this.entries = new Map(entries.map((entry) => [entry.sessionId, entry]));
|
||||
this.builtAt = entries.length > 0 ? Date.now() : 0;
|
||||
// A fresh boot plan supersedes anything an in-flight restore still holds.
|
||||
this.parked.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* The entries a viewer may see, newest plan first-come order preserved.
|
||||
*
|
||||
* @param canAccess Ownership predicate, so a user sees their own entries and
|
||||
* an admin sees all. Applied here rather than in the route so the count the
|
||||
* banner shows and the entries a click spends come from one filter.
|
||||
*/
|
||||
list(canAccess: (owner: string | undefined) => boolean): RebootRestoreEntry[] {
|
||||
this.dropIfExpired();
|
||||
return [...this.entries.values()].filter((entry) => canAccess(entry.owner));
|
||||
}
|
||||
|
||||
/**
|
||||
* Remove and return the entries a click is about to spend.
|
||||
*
|
||||
* Synchronous and total: an entry leaves the plan here, before any pane is
|
||||
* created, so a second click finds nothing to spend. Entries a caller may not
|
||||
* access are left in place, and unknown ids are ignored.
|
||||
*
|
||||
* @param sessionIds The ids to spend, or undefined for every visible entry.
|
||||
*/
|
||||
take(
|
||||
canAccess: (owner: string | undefined) => boolean,
|
||||
sessionIds: readonly string[] | undefined,
|
||||
spender: string | undefined
|
||||
): RebootRestoreEntry[] {
|
||||
this.dropIfExpired();
|
||||
const wanted = sessionIds ? new Set(sessionIds) : undefined;
|
||||
const taken: RebootRestoreEntry[] = [];
|
||||
for (const entry of [...this.entries.values()]) {
|
||||
if (wanted && !wanted.has(entry.sessionId)) continue;
|
||||
if (!canAccess(entry.owner)) continue;
|
||||
this.entries.delete(entry.sessionId);
|
||||
// Parked rather than forgotten: until this restore resolves the entry, a
|
||||
// dismiss still has to be able to reach and cancel it.
|
||||
this.parked.set(entry.sessionId, { entry, spender });
|
||||
taken.push(entry);
|
||||
}
|
||||
return taken;
|
||||
}
|
||||
|
||||
/**
|
||||
* Put entries back after a rebuild never got as far as creating a pane.
|
||||
*
|
||||
* Used for the click-time rejections that may resolve themselves: a workspace
|
||||
* that comes back, a capacity limit the user makes room under, a CLI that
|
||||
* starts once its binary is on the PATH. A conversation the user resumed by
|
||||
* hand is NOT put back, because that one cannot stop being true, and an entry
|
||||
* the banner keeps re-offering forever is noise only Dismiss can clear.
|
||||
*/
|
||||
releaseFlight(spender: string | undefined, keep: readonly RebootRestoreEntry[]): void {
|
||||
const wanted = new Set(keep.map((entry) => entry.sessionId));
|
||||
let added = 0;
|
||||
for (const [sessionId, held] of [...this.parked]) {
|
||||
if (held.spender !== spender) continue;
|
||||
this.parked.delete(sessionId);
|
||||
// Still parked means nothing cancelled it while the restore ran. A dismiss,
|
||||
// an expiry or a fresh boot plan removes it from `parked`, and then it does
|
||||
// not come back however the restore ended.
|
||||
if (wanted.has(sessionId)) {
|
||||
this.entries.set(sessionId, held.entry);
|
||||
added += 1;
|
||||
}
|
||||
}
|
||||
if (added > 0 && this.builtAt === 0) this.builtAt = Date.now();
|
||||
}
|
||||
|
||||
/** Drop the entries a viewer can see. Returns how many went. */
|
||||
clear(canAccess: (owner: string | undefined) => boolean): number {
|
||||
const removable = [...this.entries.values()].filter((entry) => canAccess(entry.owner));
|
||||
for (const entry of removable) this.entries.delete(entry.sessionId);
|
||||
// Entries a restore is holding are dismissed by the same rule, so a dismiss
|
||||
// that lands mid-restore wins. Judged on the ENTRY's owner, exactly as above,
|
||||
// rather than on who happens to be restoring it.
|
||||
let parkedRemoved = 0;
|
||||
for (const [sessionId, held] of [...this.parked]) {
|
||||
if (!canAccess(held.entry.owner)) continue;
|
||||
this.parked.delete(sessionId);
|
||||
parkedRemoved += 1;
|
||||
}
|
||||
if (this.entries.size === 0) this.builtAt = 0;
|
||||
return removable.length + parkedRemoved;
|
||||
}
|
||||
|
||||
/**
|
||||
* Claim the right to run a restore for one owner, or report that owner already
|
||||
* has one running. Callers that get `true` must call `endSpending()` in a
|
||||
* `finally` with the same owner.
|
||||
*/
|
||||
beginSpending(owner?: string): boolean {
|
||||
if (this.spending.has(owner)) return false;
|
||||
this.spending.add(owner);
|
||||
return true;
|
||||
}
|
||||
|
||||
endSpending(owner?: string): void {
|
||||
this.spending.delete(owner);
|
||||
}
|
||||
|
||||
/** Test hook: forget everything, including the single-flight claim. */
|
||||
reset(): void {
|
||||
this.entries.clear();
|
||||
this.parked.clear();
|
||||
this.builtAt = 0;
|
||||
this.spending.clear();
|
||||
}
|
||||
|
||||
private dropIfExpired(): void {
|
||||
if (this.builtAt > 0 && Date.now() - this.builtAt > PLAN_TTL_MS) {
|
||||
// A restore that took entries just before the expiry must not hand them
|
||||
// back afterwards and give an expired plan another full day of life.
|
||||
this.parked.clear();
|
||||
this.entries.clear();
|
||||
this.builtAt = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Process-wide singleton, mirroring `approvalInbox`. */
|
||||
export const rebootRestoreRegistry = new RebootRestoreRegistry();
|
||||
@@ -16,16 +16,52 @@
|
||||
|
||||
import type { FastifyInstance, FastifyRequest } from 'fastify';
|
||||
import { ApiErrorCode, createErrorResponse, type ApiResponse } from '../../types.js';
|
||||
import { isAdmin, parseBody } from '../route-helpers.js';
|
||||
import { isAdmin, parseBody, readJsonConfig, SETTINGS_PATH } from '../route-helpers.js';
|
||||
import { isMultiUserMode } from '../../config/multiuser.js';
|
||||
import { getDataDir } from '../../config/instance.js';
|
||||
import { isBlockedWebviewUrl } from '../webview-egress-policy.js';
|
||||
import { egressBlockedReason, webviewFetch } from '../webview-egress.js';
|
||||
import { CustomModelHostSchema } from '../schemas.js';
|
||||
import { readCustomModelHosts, writeCustomModelHosts, type CustomModelHost } from '../../custom-model-hosts.js';
|
||||
import type { CliEntry } from '../../config/cli-registry/types.js';
|
||||
|
||||
const CODEMAN_CONFIG_DIR = getDataDir();
|
||||
const DISCOVER_TIMEOUT_MS = 8000;
|
||||
const PROPS_TIMEOUT_MS = 5000;
|
||||
|
||||
/**
|
||||
* Claude Code's own system prompt + tool schemas cost roughly this many tokens on EVERY
|
||||
* request, before a single character of conversation history — confirmed live, twice, on
|
||||
* requests reporting `in:0 out:0` (the very first exchange) failing at ~36.4K tokens. No
|
||||
* `CLAUDE_CODE_MAX_CONTEXT_TOKENS` value fixes this: that setting only changes when Claude
|
||||
* Code decides to COMPACT conversation history, and there is no history yet on the first
|
||||
* message for it to trim. A model whose real context is below this floor will refuse
|
||||
* Claude Code's very first message outright, unconditionally.
|
||||
*
|
||||
* Set well above the ~36.4K actually measured — CLAUDE.md size, active MCP servers, and
|
||||
* enabled skills all add to a project's real baseline, so the observed figure is a floor
|
||||
* for THAT one workspace, not a ceiling for every one. Erring conservative here means a
|
||||
* borderline-safe model still gets warned about (the user can launch anyway), rather than
|
||||
* this floor missing a genuinely-too-small one because a smaller test project happened to
|
||||
* fit.
|
||||
*/
|
||||
export const CLAUDE_MIN_SAFE_CONTEXT_TOKENS = 40000;
|
||||
|
||||
/**
|
||||
* True when applying this model to this CLI is heading for a guaranteed first-message
|
||||
* failure per `CLAUDE_MIN_SAFE_CONTEXT_TOKENS` above. Gated on `contextLengthVar` (today,
|
||||
* only claude's registry entry declares one) rather than a hardcoded mode check: a CLI
|
||||
* with a small enough baseline of its own to never trip this would have no reason to
|
||||
* declare the field in the first place, so the check simply never applies to it.
|
||||
*/
|
||||
export function exceedsSafeContextFloor(
|
||||
entry: Pick<CliEntry, 'capabilities'>,
|
||||
contextLength: number | undefined
|
||||
): boolean {
|
||||
const cap = entry.capabilities.customModelInjection;
|
||||
if (cap.kind !== 'env' || !cap.contextLengthVar) return false;
|
||||
return typeof contextLength === 'number' && contextLength < CLAUDE_MIN_SAFE_CONTEXT_TOKENS;
|
||||
}
|
||||
|
||||
function adminOnly(req: FastifyRequest, reply: { code: (n: number) => unknown }): ApiResponse<never> | null {
|
||||
if (!isMultiUserMode() || isAdmin(req)) return null;
|
||||
@@ -33,7 +69,58 @@ function adminOnly(req: FastifyRequest, reply: { code: (n: number) => unknown })
|
||||
return createErrorResponse(ApiErrorCode.FORBIDDEN, 'Admin only in multi-user mode');
|
||||
}
|
||||
|
||||
async function discoverModels(host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>): Promise<string[]> {
|
||||
/**
|
||||
* `defaultModelId` names the model the Run-menu picker applies for this endpoint with
|
||||
* no further choice, so it must actually be one of the discovered `models` — a schema
|
||||
* `.refine()` can't see across the two fields the way this can, and would also run on
|
||||
* every unrelated field edit rather than only when either of these two changes.
|
||||
*/
|
||||
function invalidDefaultModel(host: Pick<CustomModelHost, 'defaultModelId' | 'models'>): ApiResponse<never> | null {
|
||||
if (host.defaultModelId === undefined) return null;
|
||||
if ((host.models ?? []).includes(host.defaultModelId)) return null;
|
||||
return createErrorResponse(
|
||||
ApiErrorCode.INVALID_INPUT,
|
||||
'defaultModelId must be one of the endpoint’s discovered models'
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Never hand the stored credential back to the browser, on GET, POST or PUT
|
||||
* alike — the file is written 0600 precisely because it holds one. `apiKeySet`
|
||||
* is what lets the editor say "unchanged if left blank" without the client
|
||||
* ever holding the real value: `applyStoredApiKey()` below is the other half,
|
||||
* treating an absent key on PUT as "keep the stored one" rather than clearing
|
||||
* it, which is what makes never returning it survivable for the edit flow.
|
||||
*/
|
||||
function redactApiKey(host: CustomModelHost): Omit<CustomModelHost, 'apiKey'> & { apiKeySet: boolean } {
|
||||
const { apiKey, ...rest } = host;
|
||||
return { ...rest, apiKeySet: !!apiKey };
|
||||
}
|
||||
|
||||
/**
|
||||
* A PUT body with no `apiKey` (or a blank one) means "leave it alone", never
|
||||
* "clear it": the editor never receives the real value to resend deliberately
|
||||
* unchanged (see redactApiKey), so the only way it can tell the two apart is
|
||||
* by omission. There is deliberately no way to CLEAR a key back to unset this
|
||||
* way — a pre-existing limitation, not something this changes.
|
||||
*/
|
||||
function applyStoredApiKey(incoming: CustomModelHost, existing: CustomModelHost): CustomModelHost {
|
||||
return incoming.apiKey ? incoming : { ...incoming, apiKey: existing.apiKey };
|
||||
}
|
||||
|
||||
/**
|
||||
* `modelContextLengths`/`modelSizesGB` are server-populated by discovery, never
|
||||
* user-entered, and PUT replaces the whole record — so merge them back in from the
|
||||
* stored host rather than trust whatever the editor's body carried (or omitted).
|
||||
* The editor only ever sends `models`/`lastDiscoveredAt` verbatim from its cached
|
||||
* copy; requiring it to also round-trip these two is exactly the kind of thing a
|
||||
* future caller forgets, same class of bug `applyStoredApiKey` exists to prevent.
|
||||
*/
|
||||
function applyDiscoveredFields(incoming: CustomModelHost, existing: CustomModelHost): CustomModelHost {
|
||||
return { ...incoming, modelContextLengths: existing.modelContextLengths, modelSizesGB: existing.modelSizesGB };
|
||||
}
|
||||
|
||||
function authHeaders(host: Pick<CustomModelHost, 'apiKey' | 'authStyle'>): Record<string, string> {
|
||||
const headers: Record<string, string> = {};
|
||||
const apiKey = host.apiKey?.trim();
|
||||
// Exactly ONE header, never both — see custom-model-hosts.ts's CustomModelAuthStyle
|
||||
@@ -41,14 +128,137 @@ async function discoverModels(host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' |
|
||||
const style = host.authStyle ?? 'bearer';
|
||||
if (apiKey && style === 'bearer') headers.Authorization = `Bearer ${apiKey}`;
|
||||
if (apiKey && style === 'api-key') headers['api-key'] = apiKey;
|
||||
return headers;
|
||||
}
|
||||
|
||||
export interface DiscoveryResult {
|
||||
models: string[];
|
||||
/** See `CustomModelHost.modelContextLengths` — only ever populated for models already loaded. */
|
||||
contextLengths: Record<string, number>;
|
||||
/** See `CustomModelHost.modelSizesGB` — populated for every model whose own listing states one. */
|
||||
sizesGB: Record<string, number>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort: pulls a file size in GB out of a model's own `description`, when the
|
||||
* server states one. llama-swap writes `"Auto-discovered 16.35 GB - parameters
|
||||
* auto-fitted by llama.cpp"` for a model it found on disk itself; a hand-configured
|
||||
* profile's own description (e.g. `"General-purpose reasoning model, MoE CPU-offloaded."`)
|
||||
* has no such figure and correctly yields no estimate rather than a guess — there is no
|
||||
* separate "give me the file size" endpoint to fall back on.
|
||||
*/
|
||||
function parseSizeGB(description: unknown): number | undefined {
|
||||
if (typeof description !== 'string') return undefined;
|
||||
const match = /(\d+(?:\.\d+)?)\s*GB\b/i.exec(description);
|
||||
if (!match) return undefined;
|
||||
const size = Number(match[1]);
|
||||
return Number.isFinite(size) && size > 0 ? size : undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort: fetches `GET /props?model=<id>` (llama.cpp-native, llama-swap-proxied) for
|
||||
* ONE already-loaded model and pulls its real `n_ctx` out. Never called for a model that
|
||||
* isn't already loaded — see the caller and `CustomModelHost.modelContextLengths` for why
|
||||
* that's a hard safety requirement, not just a nicety: llama-swap treats this endpoint's
|
||||
* `?model=` as a routing hint, and asking it about an unloaded model risks triggering an
|
||||
* actual (slow, GPU-swapping) load as a side effect of what should be read-only discovery.
|
||||
* Any failure (unreachable, non-2xx, missing/malformed field) is swallowed — one model's
|
||||
* context length is a nice-to-have, never worth failing the whole discovery pass over.
|
||||
*
|
||||
* ⚠️ FALLBACK ONLY — confirmed live to be actively WRONG for a `--fit-ctx`-launched llama-
|
||||
* swap backend: `/props`'s `n_ctx` read 154112 for a model llama-swap itself had launched
|
||||
* with `--fit-ctx 16384` (visible in `/running`'s own `cmd`), and the real server then
|
||||
* refused a request at the real 16384-token limit — `n_ctx` here appears to report the
|
||||
* model's theoretical/trained maximum, not the runtime-configured one. `parseCtxFromCmd`
|
||||
* (below), which reads the actual launch flag `/running` reports, is the primary source;
|
||||
* this is only used when that parse comes up empty (no recognized flag in `cmd`, or `cmd`
|
||||
* itself unavailable).
|
||||
*/
|
||||
async function fetchContextLength(
|
||||
host: Pick<CustomModelHost, 'baseUrl'>,
|
||||
modelId: string,
|
||||
headers: Record<string, string>
|
||||
): Promise<number | undefined> {
|
||||
try {
|
||||
const url = new URL(`${host.baseUrl.replace(/\/+$/, '')}/props`);
|
||||
url.searchParams.set('model', modelId);
|
||||
const res = await webviewFetch(url, { headers, signal: AbortSignal.timeout(PROPS_TIMEOUT_MS) });
|
||||
if (!res.ok) return undefined;
|
||||
const body = (await res.json()) as { n_ctx?: unknown; default_generation_settings?: { n_ctx?: unknown } };
|
||||
const nCtx = body.n_ctx ?? body.default_generation_settings?.n_ctx;
|
||||
return typeof nCtx === 'number' && Number.isFinite(nCtx) && nCtx > 0 ? nCtx : undefined;
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Parses the REAL configured context size out of llama-swap's own launch command for a
|
||||
* model (`/running`'s `cmd` field, e.g. `"llama-server -m ... --fit-ctx 16384 ..."`) —
|
||||
* the primary source for `modelContextLengths`, preferred over `/props`'s `n_ctx` (see
|
||||
* `fetchContextLength`'s own doc comment for why that field is unreliable here). Checks
|
||||
* `--fit-ctx` first (llama-swap's own auto-fit flag), then the plain llama.cpp
|
||||
* `-c`/`--ctx-size`/`--ctx_size` flags a hand-written launch command might use instead.
|
||||
* Returns `undefined` when `cmd` has none of these — not every launch command needs to
|
||||
* state one explicitly (llama.cpp has its own default), and guessing one would be worse
|
||||
* than the "no override applied" the caller already treats an unknown length as.
|
||||
*/
|
||||
function parseCtxFromCmd(cmd: unknown): number | undefined {
|
||||
if (typeof cmd !== 'string') return undefined;
|
||||
const match = /--fit-ctx\s+(\d+)/.exec(cmd) ?? /(?:^|\s)(?:-c|--ctx-size|--ctx_size)\s+(\d+)/.exec(cmd);
|
||||
if (!match) return undefined;
|
||||
const value = Number(match[1]);
|
||||
return Number.isFinite(value) && value > 0 ? value : undefined;
|
||||
}
|
||||
|
||||
async function discoverModels(
|
||||
host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>
|
||||
): Promise<DiscoveryResult> {
|
||||
const headers = authHeaders(host);
|
||||
const res = await webviewFetch(new URL(`${host.baseUrl.replace(/\/+$/, '')}/v1/models`), {
|
||||
headers,
|
||||
signal: AbortSignal.timeout(DISCOVER_TIMEOUT_MS),
|
||||
});
|
||||
if (!res.ok) throw new Error(`HTTP ${res.status}`);
|
||||
const body = (await res.json()) as { data?: Array<{ id?: unknown }> };
|
||||
return (body.data ?? []).map((m) => m.id).filter((id): id is string => typeof id === 'string' && id.length > 0);
|
||||
const body = (await res.json()) as {
|
||||
data?: Array<{ id?: unknown; status?: { value?: unknown }; description?: unknown }>;
|
||||
};
|
||||
const entries = body.data ?? [];
|
||||
const models = entries.map((m) => m.id).filter((id): id is string => typeof id === 'string' && id.length > 0);
|
||||
|
||||
const sizesGB: Record<string, number> = {};
|
||||
for (const entry of entries) {
|
||||
if (typeof entry.id !== 'string' || !entry.id) continue;
|
||||
const size = parseSizeGB(entry.description);
|
||||
if (size !== undefined) sizesGB[entry.id] = size;
|
||||
}
|
||||
|
||||
// llama-swap-specific, feature-detected: a server that never mentions `status` on ANY
|
||||
// entry gets no context-length enrichment at all, rather than treating "no status field"
|
||||
// as "assume unloaded" — either reading is a guess, and skipping is the safe one, since
|
||||
// fetchContextLength must only ever run against a model this server itself calls loaded.
|
||||
const hasStatusField = entries.some((m) => m && typeof m === 'object' && 'status' in m);
|
||||
const contextLengths: Record<string, number> = {};
|
||||
if (hasStatusField) {
|
||||
const loadedIds = entries
|
||||
.filter((m) => m.status && typeof m.status === 'object' && (m.status as { value?: unknown }).value === 'loaded')
|
||||
.map((m) => m.id)
|
||||
.filter((id): id is string => typeof id === 'string' && id.length > 0);
|
||||
if (loadedIds.length > 0) {
|
||||
// Primary source: the REAL launch command (see parseCtxFromCmd's own doc comment
|
||||
// for why /props's n_ctx cannot be trusted here). One /running call covers every
|
||||
// loaded model, so this never costs more requests than the old /props-only path did
|
||||
// when the cmd parse succeeds, and exactly one extra when it has to fall back.
|
||||
const swapStatus = await getLlamaSwapStatus(host);
|
||||
const cmdById = new Map(swapStatus.running.map((r) => [r.model, r.cmd]));
|
||||
for (const id of loadedIds) {
|
||||
const fromCmd = parseCtxFromCmd(cmdById.get(id));
|
||||
const ctx = fromCmd ?? (await fetchContextLength(host, id, headers));
|
||||
if (ctx !== undefined) contextLengths[id] = ctx;
|
||||
}
|
||||
}
|
||||
}
|
||||
return { models, contextLengths, sizesGB };
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -66,41 +276,498 @@ function describeFetchError(err: unknown): string {
|
||||
return message;
|
||||
}
|
||||
|
||||
export function registerCustomModelRoutes(app: FastifyInstance): void {
|
||||
app.get('/api/model-endpoints', async (req) =>
|
||||
isMultiUserMode() && !isAdmin(req) ? [] : readCustomModelHosts(CODEMAN_CONFIG_DIR)
|
||||
);
|
||||
type RedactedHost = ReturnType<typeof redactApiKey>;
|
||||
|
||||
app.post('/api/model-endpoints', async (req, reply): Promise<ApiResponse<{ host: CustomModelHost }>> => {
|
||||
/**
|
||||
* Merges a fresh `GET /v1/models` result into a host record: stamps
|
||||
* `lastDiscoveredAt`, and drops `defaultModelId` if it no longer appears in
|
||||
* the fresh list (it would otherwise leave the Run-menu picker applying a
|
||||
* model id the endpoint just told us it doesn't serve). Pure — no IO, so the
|
||||
* manual route (which reports a fetch failure's *reason* to the caller) and
|
||||
* the periodic sweep below (which only cares whether it can move on) can
|
||||
* each do their own `discoverModels()` + error handling around one shared
|
||||
* "how to apply a successful result" step.
|
||||
*/
|
||||
const RUNNING_TIMEOUT_MS = 5000;
|
||||
|
||||
export interface LlamaSwapRunningModel {
|
||||
model: string;
|
||||
state: string;
|
||||
/** The actual launch command llama-swap started this backend with, when it says one —
|
||||
* see `parseCtxFromCmd`, which reads the real configured context size out of this. */
|
||||
cmd?: string;
|
||||
}
|
||||
|
||||
export interface LlamaSwapStatus {
|
||||
/**
|
||||
* Feature-detected via `GET /running`: true only when the server answered with
|
||||
* llama-swap's own shape (`{ running: [...] }`). Plain llama.cpp (and any other
|
||||
* OpenAI-compatible server) has no such endpoint and always runs the single model
|
||||
* it was started with, so there is no "current model" to conflict with — every
|
||||
* caller must treat `isLlamaSwap: false` as "nothing to check", never as an error.
|
||||
*/
|
||||
isLlamaSwap: boolean;
|
||||
running: LlamaSwapRunningModel[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Distinguishes llama-swap from a plain llama.cpp/OpenAI-compatible server, and reports
|
||||
* what llama-swap currently has loaded — llama.cpp only ever runs one GGUF at a time, and
|
||||
* llama-swap unloads/reloads it on demand when a request asks for a different one, which
|
||||
* can take anywhere from a few seconds to over a minute. Read-only: this never triggers a
|
||||
* swap itself (unlike `/props?model=`, `/running` takes no `model` parameter to route by).
|
||||
* Best-effort like `discoverModels()`'s siblings: any failure (unreachable, non-2xx,
|
||||
* unexpected shape) reads as "not llama-swap", never thrown.
|
||||
*/
|
||||
export async function getLlamaSwapStatus(
|
||||
host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>
|
||||
): Promise<LlamaSwapStatus> {
|
||||
try {
|
||||
const res = await webviewFetch(new URL(`${host.baseUrl.replace(/\/+$/, '')}/running`), {
|
||||
headers: authHeaders(host),
|
||||
signal: AbortSignal.timeout(RUNNING_TIMEOUT_MS),
|
||||
});
|
||||
if (!res.ok) return { isLlamaSwap: false, running: [] };
|
||||
const body = (await res.json()) as { running?: unknown };
|
||||
if (!Array.isArray(body.running)) return { isLlamaSwap: false, running: [] };
|
||||
const running = body.running
|
||||
.filter(
|
||||
(r): r is { model: string; state?: unknown; cmd?: unknown } =>
|
||||
!!r && typeof r === 'object' && typeof (r as { model?: unknown }).model === 'string'
|
||||
)
|
||||
.map((r) => ({
|
||||
model: r.model,
|
||||
state: typeof r.state === 'string' ? r.state : 'unknown',
|
||||
cmd: typeof r.cmd === 'string' ? r.cmd : undefined,
|
||||
}));
|
||||
return { isLlamaSwap: true, running };
|
||||
} catch {
|
||||
return { isLlamaSwap: false, running: [] };
|
||||
}
|
||||
}
|
||||
|
||||
interface LlamaSwapLogTail {
|
||||
latestLine?: string;
|
||||
lastAccessedAt: number;
|
||||
controller: AbortController;
|
||||
}
|
||||
|
||||
/** One open `/api/events` tail per endpoint, keyed by host id — see `getLatestLlamaSwapLogLine`. */
|
||||
const llamaSwapLogTails = new Map<string, LlamaSwapLogTail>();
|
||||
|
||||
/** A tail nothing has asked about in this long is closed by the next `pruneIdleLlamaSwapLogTails` sweep. */
|
||||
const LOG_TAIL_IDLE_MS = 30_000;
|
||||
|
||||
/**
|
||||
* Parses one `data: {...}` payload from llama-swap's `GET /api/events` SSE stream and
|
||||
* returns the backend (never llama-swap's own proxy) log text it carries, or `undefined`
|
||||
* for anything else (a different event `type`, a malformed frame, a proxy-sourced one).
|
||||
*
|
||||
* The real shape, confirmed live against a real llama-swap deployment — NOT documented
|
||||
* anywhere the plan doc's original research found, and genuinely surprising the first
|
||||
* time around: `GET /logs` (the endpoint that name suggests, and this feature's own
|
||||
* first cut was built against) turns out to carry ONLY llama-swap's own proxy
|
||||
* request-access log — it never once showed a single backend line even seconds after a
|
||||
* real, confirmed model swap. The backend llama-server process's actual stdout
|
||||
* (`load_model: ...`, `llama_server: model loaded`) only ever showed up in `/api/events`,
|
||||
* as `{"type":"logData","data":"<JSON-string>"}` whose OWN `data` field parses to a
|
||||
* second object, `{"data": "<newline-joined log text>", "source": "proxy" | "upstream"}`
|
||||
* — `source` is the exact, explicit distinguisher (`upstream` = the backend process,
|
||||
* `proxy` = llama-swap's own line), not a guessed regex against the text itself.
|
||||
*/
|
||||
function parseBackendLogDataEvent(dataLine: string): string | undefined {
|
||||
let outer: unknown;
|
||||
try {
|
||||
outer = JSON.parse(dataLine);
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
if (
|
||||
!outer ||
|
||||
typeof outer !== 'object' ||
|
||||
(outer as { type?: unknown }).type !== 'logData' ||
|
||||
typeof (outer as { data?: unknown }).data !== 'string'
|
||||
) {
|
||||
return undefined;
|
||||
}
|
||||
let inner: unknown;
|
||||
try {
|
||||
inner = JSON.parse((outer as { data: string }).data);
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
if (
|
||||
!inner ||
|
||||
typeof inner !== 'object' ||
|
||||
(inner as { source?: unknown }).source !== 'upstream' ||
|
||||
typeof (inner as { data?: unknown }).data !== 'string'
|
||||
) {
|
||||
return undefined;
|
||||
}
|
||||
return (inner as { data: string }).data;
|
||||
}
|
||||
|
||||
/**
|
||||
* Reads `GET /api/events` forever (until `entry.controller` aborts it), updating
|
||||
* `entry.latestLine` with the most recent BACKEND log line seen (see
|
||||
* `parseBackendLogDataEvent`). Fire-and-forget: the caller never awaits this — it runs
|
||||
* for the tail's whole lifetime in the background, and `getLatestLlamaSwapLogLine` just
|
||||
* reads whatever `entry.latestLine` currently holds. SSE frames are separated by a blank
|
||||
* line (`\n\n`), buffered the same way `/running`'s NDJSON-shaped siblings buffer partial
|
||||
* chunks — a frame split across two `reader.read()` calls must not be parsed early.
|
||||
*/
|
||||
/**
|
||||
* Cap on the unparsed remainder held between reads of the backend log stream. One
|
||||
* SSE frame is a status line, so this is orders of magnitude more than a real frame
|
||||
* needs; it exists so a server that never emits a frame boundary cannot grow the
|
||||
* buffer without bound for the life of the connection.
|
||||
*/
|
||||
const MAX_LOG_TAIL_BUFFER_CHARS = 64 * 1024;
|
||||
|
||||
async function pumpLlamaSwapLogTail(
|
||||
host: Pick<CustomModelHost, 'id' | 'baseUrl' | 'apiKey' | 'authStyle'>,
|
||||
entry: LlamaSwapLogTail
|
||||
): Promise<void> {
|
||||
try {
|
||||
const res = await webviewFetch(new URL(`${host.baseUrl.replace(/\/+$/, '')}/api/events`), {
|
||||
headers: authHeaders(host),
|
||||
signal: entry.controller.signal,
|
||||
});
|
||||
if (!res.ok || !res.body) return;
|
||||
const reader = res.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let buffer = '';
|
||||
for (;;) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
buffer += decoder.decode(value, { stream: true });
|
||||
const frames = buffer.split('\n\n');
|
||||
buffer = frames.pop() ?? '';
|
||||
// The remainder only shrinks at a frame boundary, so a server that streams
|
||||
// without `\n\n` (or one very long frame) would grow it for as long as the
|
||||
// connection is held, which is indefinitely by design. Past the cap the
|
||||
// partial frame cannot become a useful log line anyway, so drop it and
|
||||
// resynchronise on the next boundary rather than buffering forever.
|
||||
if (buffer.length > MAX_LOG_TAIL_BUFFER_CHARS) buffer = '';
|
||||
for (const frame of frames) {
|
||||
const dataLine = frame.split('\n').find((l) => l.startsWith('data:'));
|
||||
if (!dataLine) continue;
|
||||
const backendText = parseBackendLogDataEvent(dataLine.slice('data:'.length));
|
||||
if (!backendText) continue;
|
||||
const lines = backendText.split('\n').filter((l) => l.trim());
|
||||
if (lines.length > 0) entry.latestLine = lines[lines.length - 1]!.trim();
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
// connection dropped / aborted / endpoint unreachable — a future access starts fresh
|
||||
} finally {
|
||||
// Delete by IDENTITY, not just by key: an aborted pump can finish after a NEWER
|
||||
// entry was already created for the same endpoint id (e.g. abort-then-immediately-
|
||||
// re-request), and deleting unconditionally would remove that newer entry and orphan
|
||||
// its connection — nothing would ever prune it, since pruneIdleLlamaSwapLogTails only
|
||||
// walks entries still present in the map.
|
||||
if (llamaSwapLogTails.get(host.id) === entry) {
|
||||
llamaSwapLogTails.delete(host.id);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Real-time "what is llama.cpp actually doing right now" for the loading banner
|
||||
* (docs/custom-model-endpoints-plan.md): llama-swap's `GET /api/events` SSE stream
|
||||
* carries the backend llama-server process's own stdout — `load_model: loading model
|
||||
* '<path>'`, `load_model: initializing, n_slots = N, n_ctx_slot = N`, `llama_server:
|
||||
* model loaded`, etc — tagged `source: "upstream"`, distinct from llama-swap's own
|
||||
* `source: "proxy"` request-access lines (see `parseBackendLogDataEvent`). Confirmed
|
||||
* live against a real llama-swap deployment, including through an actual forced model
|
||||
* swap end-to-end.
|
||||
*
|
||||
* Held OPEN per endpoint rather than re-opened on every 1s poll — confirmed live to stay
|
||||
* open indefinitely (read past 220KB over 8 seconds with no `done`), unlike `/logs`
|
||||
* (see `parseBackendLogDataEvent`'s doc comment), so reconnecting each poll would be
|
||||
* pure waste. One connection is reused across every session currently watching a load on
|
||||
* that endpoint; since llama.cpp/llama-swap only ever runs one model at a time, a line
|
||||
* seen while a load is in flight is safe to attribute to that load (a deployment that
|
||||
* could load several models concurrently would need a per-model tag this format doesn't
|
||||
* provide).
|
||||
*
|
||||
* Lazily started on first access and idle-closed rather than left open forever — see
|
||||
* `pruneIdleLlamaSwapLogTails`.
|
||||
*/
|
||||
export function getLatestLlamaSwapLogLine(
|
||||
host: Pick<CustomModelHost, 'id' | 'baseUrl' | 'apiKey' | 'authStyle'>
|
||||
): string | undefined {
|
||||
let entry = llamaSwapLogTails.get(host.id);
|
||||
if (!entry) {
|
||||
entry = { lastAccessedAt: Date.now(), controller: new AbortController() };
|
||||
llamaSwapLogTails.set(host.id, entry);
|
||||
void pumpLlamaSwapLogTail(host, entry);
|
||||
}
|
||||
entry.lastAccessedAt = Date.now();
|
||||
return entry.latestLine;
|
||||
}
|
||||
|
||||
/**
|
||||
* Closes EVERY open log tail. The idle sweep above only runs on server.ts's periodic
|
||||
* interval, and that interval is disposed on shutdown, so without this an outbound
|
||||
* stream outlives `WebServer.stop()` against CLAUDE.md's "clear Maps in stop()" rule.
|
||||
* Harmless today only because `cli.ts`'s shutdown handler reaches `process.exit(0)`,
|
||||
* which is not a property to rely on: tests and any in-process restart do not.
|
||||
*/
|
||||
export function closeAllLlamaSwapLogTails(): void {
|
||||
for (const entry of llamaSwapLogTails.values()) entry.controller.abort();
|
||||
llamaSwapLogTails.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* Closes any log tail nothing has called `getLatestLlamaSwapLogLine` about in
|
||||
* `LOG_TAIL_IDLE_MS` — a stream nobody is polling is an open connection with nothing to
|
||||
* show for it. Called from the same periodic sweep as `detectCustomModelSwapDisplacements`
|
||||
* in server.ts, not its own timer.
|
||||
*/
|
||||
export function pruneIdleLlamaSwapLogTails(now = Date.now()): void {
|
||||
for (const [id, entry] of llamaSwapLogTails) {
|
||||
if (now - entry.lastAccessedAt > LOG_TAIL_IDLE_MS) {
|
||||
entry.controller.abort();
|
||||
llamaSwapLogTails.delete(id);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Actually kicks off llama-swap's lazy model load, rather than waiting for the launched
|
||||
* CLI's own first prompt to do it. llama-swap has no separate "switch model" admin
|
||||
* endpoint — the ONLY thing that starts a swap is a real inference request naming the
|
||||
* model (confirmed live: applying a selection alone never appeared in the llama-swap
|
||||
* server's own logs; nothing had actually asked it to load anything). This sends the
|
||||
* smallest real request that will — `max_tokens: 1`, one throwaway user message — to
|
||||
* `${baseUrl}/v1/chat/completions`, the OpenAI-compatible endpoint every supported
|
||||
* harness already points at.
|
||||
*
|
||||
* Deliberately fire-and-forget: the caller (the apply/create routes) returns to the
|
||||
* client immediately, and the frontend's own polling (`GET .../running-status`) is what
|
||||
* actually confirms readiness — this call's response is never read, just its side
|
||||
* effect. No abort/timeout of its own either: a real load can take well over a minute for
|
||||
* a large model, and this is a normal long-running Node process, so there is nothing to
|
||||
* clean up by cutting it short. Errors are swallowed for the same reason `discoverModels`'s
|
||||
* siblings swallow theirs — one endpoint's hiccup here is a nice-to-have that failed, not
|
||||
* something worth surfacing as a request failure four layers up.
|
||||
*/
|
||||
export function triggerLlamaSwapLoad(
|
||||
host: Pick<CustomModelHost, 'baseUrl' | 'apiKey' | 'authStyle'>,
|
||||
modelId: string
|
||||
): void {
|
||||
const url = new URL(`${host.baseUrl.replace(/\/+$/, '')}/v1/chat/completions`);
|
||||
webviewFetch(url, {
|
||||
method: 'POST',
|
||||
headers: { ...authHeaders(host), 'content-type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
model: modelId,
|
||||
messages: [{ role: 'user', content: 'Hi' }],
|
||||
max_tokens: 1,
|
||||
stream: false,
|
||||
}),
|
||||
}).catch(() => {
|
||||
// best-effort — see the doc comment above
|
||||
});
|
||||
}
|
||||
|
||||
function applyDiscoveredModels(host: CustomModelHost, result: DiscoveryResult): CustomModelHost {
|
||||
const { models, contextLengths, sizesGB } = result;
|
||||
const defaultModelId = host.defaultModelId && models.includes(host.defaultModelId) ? host.defaultModelId : undefined;
|
||||
// Merge onto what's already known rather than replacing: a model not probed this round
|
||||
// (not currently loaded) keeps whatever context length an earlier round already learned
|
||||
// for it, and one no longer in the fresh list is dropped, same reasoning as defaultModelId.
|
||||
const merged = { ...host.modelContextLengths, ...contextLengths };
|
||||
const kept = Object.fromEntries(Object.entries(merged).filter(([id]) => models.includes(id)));
|
||||
const modelContextLengths = Object.keys(kept).length > 0 ? kept : undefined;
|
||||
// sizesGB, unlike contextLengths, is populated for every model in the SAME pass (no
|
||||
// loaded-only restriction — see parseSizeGB), so this is closer to a plain replace, but
|
||||
// still merges onto the previous round rather than dropping a size for a model whose
|
||||
// description happened to omit the figure on this particular pass.
|
||||
const mergedSizes = { ...host.modelSizesGB, ...sizesGB };
|
||||
const keptSizes = Object.fromEntries(Object.entries(mergedSizes).filter(([id]) => models.includes(id)));
|
||||
const modelSizesGB = Object.keys(keptSizes).length > 0 ? keptSizes : undefined;
|
||||
return {
|
||||
...host,
|
||||
models,
|
||||
defaultModelId,
|
||||
modelContextLengths,
|
||||
modelSizesGB,
|
||||
lastDiscoveredAt: new Date().toISOString(),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* `customModelEndpointsEnabled` defaults OFF (unlike `showPlanUsageLimits`'s
|
||||
* absent-means-on in `readPlanUsageTelemetryEnabled`), so mirror the frontend's
|
||||
* own gate (`session-ui.js`'s `!settings.customModelEndpointsEnabled`) rather
|
||||
* than that reader's default. Exists so the periodic re-discovery sweep in
|
||||
* server.ts can skip entirely while the feature is off, instead of polling
|
||||
* every saved endpoint forever regardless of the setting.
|
||||
*/
|
||||
export async function readCustomModelEndpointsEnabled(): Promise<boolean> {
|
||||
const settings = await readJsonConfig<Record<string, unknown>>(SETTINGS_PATH, 'settings.json', {});
|
||||
return settings.customModelEndpointsEnabled === true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-discovers every saved endpoint's models, best-effort. One endpoint being
|
||||
* unreachable (powered off, wrong network) must not stop the others from
|
||||
* refreshing, and a read-modify-write per host (rather than one batch write
|
||||
* at the end) means a crash or restart mid-sweep loses at most the endpoints
|
||||
* not yet reached, never a write already applied. Exported so both the
|
||||
* periodic timer (server.ts) and a test can drive it directly.
|
||||
*/
|
||||
export async function refreshAllCustomModelHosts(): Promise<void> {
|
||||
const dataDir = getDataDir();
|
||||
const hosts = await readCustomModelHosts(dataDir);
|
||||
for (const host of hosts) {
|
||||
if (isBlockedWebviewUrl(host.baseUrl)) continue;
|
||||
let result: DiscoveryResult;
|
||||
try {
|
||||
result = await discoverModels(host);
|
||||
} catch {
|
||||
continue; // unreachable this cycle — try again next tick, not fatal to the sweep
|
||||
}
|
||||
// Re-read + splice by id rather than reusing the array captured above: an
|
||||
// admin editing or deleting an endpoint via the API mid-sweep must win,
|
||||
// not be silently overwritten by a refresh that started before their change.
|
||||
const current = await readCustomModelHosts(dataDir);
|
||||
const index = current.findIndex((item) => item.id === host.id);
|
||||
if (index === -1) continue; // deleted mid-sweep
|
||||
current[index] = applyDiscoveredModels(current[index], result);
|
||||
await writeCustomModelHosts(dataDir, current);
|
||||
}
|
||||
}
|
||||
|
||||
/** The subset of `Session` this sweep needs — kept minimal so a test can pass a plain object. */
|
||||
export interface CustomModelSessionLike {
|
||||
id: string;
|
||||
name: string;
|
||||
customModel?: { endpointId: string; modelId: string; label?: string };
|
||||
}
|
||||
|
||||
/** One session whose model was just found evicted, ready to broadcast as `CustomModelSwappedOut`. */
|
||||
export interface CustomModelSwapDisplacement {
|
||||
sessionId: string;
|
||||
sessionName: string;
|
||||
endpointId: string;
|
||||
previousModel: string;
|
||||
currentlyLoadedModel: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Detects when a live session's own custom-model selection is no longer the model
|
||||
* llama-swap actually has loaded — evicted by ANOTHER session's activity on the same
|
||||
* endpoint, since llama.cpp/llama-swap runs one model at a time (the apply/create routes'
|
||||
* own swap-conflict check only ever runs at THAT session's own launch/apply moment, so it
|
||||
* cannot catch a later eviction triggered by a different session's normal use — confirmed
|
||||
* live: a session created while nothing else had a live conflict at that instant can still
|
||||
* get silently displaced afterward). Read-only, and best-effort per endpoint exactly like
|
||||
* `refreshAllCustomModelHosts`'s sibling sweep — one endpoint's hiccup here never blocks
|
||||
* checking the others.
|
||||
*
|
||||
* `notifiedSessionIds` is the caller's own de-dupe state (`server.ts` keeps one `Set` across
|
||||
* sweeps), mutated in place: a session id is added once displaced and removed again once its
|
||||
* own model is loaded and ready — so a LATER, genuinely new displacement can notify again
|
||||
* rather than the session staying silently un-notified forever after the first one.
|
||||
*/
|
||||
export async function detectCustomModelSwapDisplacements(
|
||||
sessions: Iterable<CustomModelSessionLike>,
|
||||
notifiedSessionIds: Set<string>
|
||||
): Promise<CustomModelSwapDisplacement[]> {
|
||||
const byEndpoint = new Map<string, CustomModelSessionLike[]>();
|
||||
for (const session of sessions) {
|
||||
if (!session.customModel) continue;
|
||||
const group = byEndpoint.get(session.customModel.endpointId);
|
||||
if (group) group.push(session);
|
||||
else byEndpoint.set(session.customModel.endpointId, [session]);
|
||||
}
|
||||
if (byEndpoint.size === 0) return [];
|
||||
|
||||
const hosts = await readCustomModelHosts(getDataDir());
|
||||
const displacements: CustomModelSwapDisplacement[] = [];
|
||||
|
||||
for (const [endpointId, group] of byEndpoint) {
|
||||
const host = hosts.find((h) => h.id === endpointId);
|
||||
if (!host) continue; // endpoint deleted since these sessions were created — nothing to check
|
||||
let status: LlamaSwapStatus;
|
||||
try {
|
||||
status = await getLlamaSwapStatus(host);
|
||||
} catch {
|
||||
continue; // unreachable this cycle — try again next tick, not fatal to the sweep
|
||||
}
|
||||
// Not llama-swap (feature-detected) or nothing loaded at all: nothing has been evicted,
|
||||
// by construction — a plain llama.cpp/OpenAI-compatible server only ever runs the one
|
||||
// model it was started with, so there is no "current model" to conflict with.
|
||||
if (!status.isLlamaSwap || status.running.length === 0) continue;
|
||||
const currentlyLoaded = status.running.find((r) => r.state === 'ready')?.model ?? status.running[0]?.model;
|
||||
if (!currentlyLoaded) continue;
|
||||
|
||||
for (const session of group) {
|
||||
const modelId = session.customModel!.modelId;
|
||||
const stillLoaded = status.running.some((r) => r.model === modelId);
|
||||
if (stillLoaded) {
|
||||
notifiedSessionIds.delete(session.id); // back to normal — a future eviction can notify again
|
||||
continue;
|
||||
}
|
||||
if (notifiedSessionIds.has(session.id)) continue; // already told them once for this displacement
|
||||
notifiedSessionIds.add(session.id);
|
||||
displacements.push({
|
||||
sessionId: session.id,
|
||||
sessionName: session.name,
|
||||
endpointId,
|
||||
previousModel: modelId,
|
||||
currentlyLoadedModel: currentlyLoaded,
|
||||
});
|
||||
}
|
||||
}
|
||||
return displacements;
|
||||
}
|
||||
|
||||
export function registerCustomModelRoutes(app: FastifyInstance): void {
|
||||
app.get('/api/model-endpoints', async (req): Promise<RedactedHost[]> => {
|
||||
if (isMultiUserMode() && !isAdmin(req)) return [];
|
||||
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
|
||||
return hosts.map(redactApiKey);
|
||||
});
|
||||
|
||||
app.post('/api/model-endpoints', async (req, reply): Promise<ApiResponse<{ host: RedactedHost }>> => {
|
||||
const denied = adminOnly(req, reply);
|
||||
if (denied) return denied;
|
||||
const host = parseBody(CustomModelHostSchema, req.body);
|
||||
if (isBlockedWebviewUrl(host.baseUrl)) {
|
||||
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
|
||||
}
|
||||
const badDefault = invalidDefaultModel(host);
|
||||
if (badDefault) return badDefault;
|
||||
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
|
||||
if (hosts.some((item) => item.id === host.id)) {
|
||||
return createErrorResponse(ApiErrorCode.ALREADY_EXISTS, 'Model endpoint already exists');
|
||||
}
|
||||
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, [...hosts, host]);
|
||||
return { success: true, data: { host } };
|
||||
return { success: true, data: { host: redactApiKey(host) } };
|
||||
});
|
||||
|
||||
app.put('/api/model-endpoints/:id', async (req, reply): Promise<ApiResponse<{ host: CustomModelHost }>> => {
|
||||
app.put('/api/model-endpoints/:id', async (req, reply): Promise<ApiResponse<{ host: RedactedHost }>> => {
|
||||
const denied = adminOnly(req, reply);
|
||||
if (denied) return denied;
|
||||
const { id } = req.params as { id: string };
|
||||
const host = parseBody(CustomModelHostSchema, { ...(req.body as object), id });
|
||||
if (isBlockedWebviewUrl(host.baseUrl)) {
|
||||
const incoming = parseBody(CustomModelHostSchema, { ...(req.body as object), id });
|
||||
if (isBlockedWebviewUrl(incoming.baseUrl)) {
|
||||
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
|
||||
}
|
||||
const badDefault = invalidDefaultModel(incoming);
|
||||
if (badDefault) return badDefault;
|
||||
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
|
||||
const index = hosts.findIndex((item) => item.id === id);
|
||||
if (index === -1) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
|
||||
const host = applyDiscoveredFields(applyStoredApiKey(incoming, hosts[index]), hosts[index]);
|
||||
const next = [...hosts];
|
||||
next[index] = host;
|
||||
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, next);
|
||||
return { success: true, data: { host } };
|
||||
return { success: true, data: { host: redactApiKey(host) } };
|
||||
});
|
||||
|
||||
app.delete('/api/model-endpoints/:id', async (req, reply): Promise<ApiResponse<{ id: string }>> => {
|
||||
@@ -129,11 +796,11 @@ export function registerCustomModelRoutes(app: FastifyInstance): void {
|
||||
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
|
||||
}
|
||||
try {
|
||||
const models = await discoverModels(host);
|
||||
const result = await discoverModels(host);
|
||||
const next = [...hosts];
|
||||
next[index] = { ...host, models, lastDiscoveredAt: new Date().toISOString() };
|
||||
next[index] = applyDiscoveredModels(host, result);
|
||||
await writeCustomModelHosts(CODEMAN_CONFIG_DIR, next);
|
||||
return { success: true, data: { models } };
|
||||
return { success: true, data: { models: result.models } };
|
||||
} catch (err) {
|
||||
const blocked = egressBlockedReason(err);
|
||||
return createErrorResponse(
|
||||
@@ -143,4 +810,38 @@ export function registerCustomModelRoutes(app: FastifyInstance): void {
|
||||
}
|
||||
}
|
||||
);
|
||||
|
||||
// Read-only, no admin gate: any session owner who can already point their own session
|
||||
// at this endpoint (POST .../custom-model, ungated by design — see session-routes.ts)
|
||||
// can equally ask what it currently has loaded, before or while that apply is pending.
|
||||
app.get(
|
||||
'/api/model-endpoints/:id/running-status',
|
||||
async (
|
||||
req
|
||||
): Promise<
|
||||
ApiResponse<{
|
||||
isLlamaSwap: boolean;
|
||||
running: Array<Pick<LlamaSwapRunningModel, 'model' | 'state'>>;
|
||||
logLine?: string;
|
||||
}>
|
||||
> => {
|
||||
const { id } = req.params as { id: string };
|
||||
const hosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
|
||||
const host = hosts.find((item) => item.id === id);
|
||||
if (!host) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
|
||||
if (isBlockedWebviewUrl(host.baseUrl)) {
|
||||
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Endpoint base URL is not allowed');
|
||||
}
|
||||
const status = await getLlamaSwapStatus(host);
|
||||
// Only worth tailing /api/events once llama-swap is actually confirmed — a plain
|
||||
// llama.cpp/OpenAI-compatible server has no such endpoint at all.
|
||||
const logLine = status.isLlamaSwap ? getLatestLlamaSwapLogLine(host) : undefined;
|
||||
// `cmd` (the literal llama-server launch line, which can carry model paths and
|
||||
// --api-key) exists only so parseCtxFromCmd() can read it server-side during
|
||||
// discovery — this un-gated, polled-every-second route has no reason to hand it
|
||||
// to the browser, which only ever reads `model`/`state`.
|
||||
const running = status.running.map(({ model, state }) => ({ model, state }));
|
||||
return { success: true, data: { isLlamaSwap: status.isLlamaSwap, running, logLine } };
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
+11
-1
@@ -11,6 +11,7 @@ export { registerCronRoutes } from './cron-routes.js';
|
||||
export { registerSystemRoutes } from './system-routes.js';
|
||||
export { registerHookEventRoutes } from './hook-event-routes.js';
|
||||
export { registerApprovalRoutes } from './approval-routes.js';
|
||||
export { registerRebootRestoreRoutes } from './reboot-restore-routes.js';
|
||||
export { registerReadMyMindRoutes } from './readmymind-routes.js';
|
||||
export { registerStatusTelemetryRoutes } from './status-telemetry-routes.js';
|
||||
export { registerCaseRoutes } from './case-routes.js';
|
||||
@@ -27,4 +28,13 @@ export { registerWsRoutes } from './ws-routes.js';
|
||||
export { registerVoiceRoutes } from './voice-routes.js';
|
||||
export { registerWebviewRoutes, tryWebviewRefererFallback } from './webview-routes.js';
|
||||
export { registerTabLayoutRoutes } from './tab-layout-routes.js';
|
||||
export { registerCustomModelRoutes } from './custom-model-routes.js';
|
||||
export {
|
||||
registerCustomModelRoutes,
|
||||
refreshAllCustomModelHosts,
|
||||
readCustomModelEndpointsEnabled,
|
||||
closeAllLlamaSwapLogTails,
|
||||
detectCustomModelSwapDisplacements,
|
||||
pruneIdleLlamaSwapLogTails,
|
||||
type CustomModelSessionLike,
|
||||
type CustomModelSwapDisplacement,
|
||||
} from './custom-model-routes.js';
|
||||
|
||||
@@ -0,0 +1,313 @@
|
||||
/**
|
||||
* @fileoverview Reboot-restore routes: offer back the sessions a host reboot destroyed.
|
||||
*
|
||||
* The boot pass leaves a plan in `web/reboot-restore-registry` when the machine
|
||||
* plausibly rebooted. The board reads it, shows a banner, and the user decides:
|
||||
* - `GET /api/reboot-restore`: what is on offer, ownership-scoped
|
||||
* - `POST /api/reboot-restore/restore`: rebuild some or all of it
|
||||
* - `POST /api/reboot-restore/dismiss`: drop the offer
|
||||
*
|
||||
* A click, not the heuristic, is what creates panes. The heuristic only decides
|
||||
* whether the banner appears, so a wrong yes costs a line of text the user
|
||||
* dismisses rather than N CLI processes nobody asked for.
|
||||
*
|
||||
* Rebuilding is take-then-build: entries leave the plan synchronously at the top
|
||||
* of the route, before the first `await`, and the whole route is single-flighted,
|
||||
* so a double-click or two devices cannot put two panes on one conversation.
|
||||
* Three things are re-checked at click time rather than trusted from boot: the
|
||||
* owner's privilege grant, the workspace still being on disk, and the
|
||||
* conversation not already being live because the user resumed it by hand.
|
||||
*
|
||||
* A rebuilt session comes back attached, idle and disarmed. Respawn controllers
|
||||
* and Ralph loops are deliberately not re-armed, and its terminal scrollback is
|
||||
* gone, because the pane is new. The banner says so.
|
||||
*/
|
||||
|
||||
import { FastifyInstance } from 'fastify';
|
||||
import { existsSync } from 'node:fs';
|
||||
import { ApiErrorCode, createErrorResponse, getErrorMessage } from '../../types.js';
|
||||
import { RebootRestoreRequestSchema } from '../schemas.js';
|
||||
import {
|
||||
parseBody,
|
||||
getAuthUser,
|
||||
canAccessOwned,
|
||||
ownerFor,
|
||||
isWorkingDirAllowedForUsername,
|
||||
sessionCapacityMessage,
|
||||
} from '../route-helpers.js';
|
||||
import { rebootRestoreRegistry } from '../reboot-restore-registry.js';
|
||||
import { rejectAlreadyLive, type RebootRestoreEntry, type RebootRestoreRejection } from '../../reboot-restore.js';
|
||||
import { clampEnvOverridesForOwner } from '../../session-env-clamp.js';
|
||||
import { Session } from '../../session.js';
|
||||
import { resolveClaudeModeForUsername } from '../../user-store.js';
|
||||
import { getCli } from '../../config/cli-registry/registry.js';
|
||||
import { applyWorkspaceHooks, seedAgentSessionPreamble } from '../../hooks-config.js';
|
||||
import { getLifecycleLog } from '../../session-lifecycle-log.js';
|
||||
import { STATS_COLLECTION_INTERVAL_MS } from '../../config/server-timing.js';
|
||||
import { SseEvent } from '../sse-events.js';
|
||||
import type { SessionAttachmentHistoryItem } from '../../types.js';
|
||||
import type { SessionPort, EventPort, ConfigPort, InfraPort } from '../ports/index.js';
|
||||
|
||||
type RebootRestoreCtx = SessionPort & EventPort & ConfigPort & InfraPort;
|
||||
|
||||
/** The banner's view of one restorable session. The record itself never leaves the server. */
|
||||
function toBannerItem(entry: RebootRestoreEntry) {
|
||||
return {
|
||||
id: entry.sessionId,
|
||||
name: entry.name,
|
||||
workingDir: entry.workingDir,
|
||||
mode: entry.mode,
|
||||
owner: entry.owner,
|
||||
};
|
||||
}
|
||||
|
||||
export function registerRebootRestoreRoutes(app: FastifyInstance, ctx: RebootRestoreCtx): void {
|
||||
const accessorFor = (req: Parameters<typeof getAuthUser>[0]) => {
|
||||
const user = getAuthUser(req);
|
||||
return (owner: string | undefined) => canAccessOwned(user, owner);
|
||||
};
|
||||
|
||||
// ========== What is on offer ==========
|
||||
|
||||
app.get('/api/reboot-restore', async (req) => {
|
||||
const entries = rebootRestoreRegistry.list(accessorFor(req));
|
||||
return {
|
||||
sessions: entries.map(toBannerItem),
|
||||
// Said plainly here so the banner never implies a full restore: the pane is
|
||||
// new, so the conversation continues and the terminal history does not.
|
||||
scrollbackRestored: false,
|
||||
};
|
||||
});
|
||||
|
||||
// ========== Spend it ==========
|
||||
|
||||
app.post('/api/reboot-restore/restore', async (req, reply) => {
|
||||
const body = parseBody(RebootRestoreRequestSchema, req.body, 'Invalid reboot restore request');
|
||||
const canAccess = accessorFor(req);
|
||||
const owner = ownerFor(req);
|
||||
|
||||
// Take BEFORE the first await: a second click must find nothing to spend.
|
||||
// The flight is per owner, because `take()` already guarantees two callers
|
||||
// never receive the same entry, so one user's restore need not block another's.
|
||||
if (!rebootRestoreRegistry.beginSpending(owner)) {
|
||||
return reply.code(409).send(createErrorResponse(ApiErrorCode.CONFLICT, 'A reboot restore is already running'));
|
||||
}
|
||||
const taken = rebootRestoreRegistry.take(canAccess, body.sessionIds, owner);
|
||||
// Entries nothing built a pane for, returned to the plan on every exit path
|
||||
// including a throw. Without this a failure between here and the loop would
|
||||
// spend the offer and rebuild nothing, and the plan cannot be rebuilt.
|
||||
const unspent = new Set(taken);
|
||||
|
||||
try {
|
||||
if (taken.length === 0) return { restored: [], skipped: [] };
|
||||
|
||||
// The plan was built at boot and the board has moved on since. A conversation
|
||||
// the user resumed by hand from the Resume list is already on screen, and a
|
||||
// second pane on it would fight the first for the same transcript. This one
|
||||
// is never re-offered: unlike a missing workspace, it cannot stop being true.
|
||||
// Read fresh each time rather than snapshotted once: the loop below awaits a
|
||||
// real `startInteractive()` per entry, so by the tenth entry a snapshot taken
|
||||
// here is tens of seconds old, and a conversation the user resumed by hand in
|
||||
// that window would be invisible to it.
|
||||
const liveSessionIds = () => new Set(ctx.sessions.keys());
|
||||
const liveConversationIds = () =>
|
||||
new Set(
|
||||
[...ctx.sessions.values()].map((session) => session.claudeSessionId).filter((id): id is string => !!id)
|
||||
);
|
||||
const { restore, skipped } = rejectAlreadyLive(taken, liveSessionIds(), liveConversationIds());
|
||||
for (const entry of taken) {
|
||||
if (skipped.some((s) => s.sessionId === entry.sessionId)) unspent.delete(entry);
|
||||
}
|
||||
|
||||
const restored: ReturnType<typeof toBannerItem>[] = [];
|
||||
const failures: RebootRestoreRejection[] = [...skipped];
|
||||
const workspaceHooksEnabled = await ctx.getWorkspaceHooksEnabled();
|
||||
|
||||
for (const entry of restore) {
|
||||
// The already-live check, re-run against the board as it is NOW. The pass
|
||||
// above decided the batch; this catches a conversation that went live while
|
||||
// an earlier entry in this same batch was starting. Spent rather than
|
||||
// returned to the plan, for the same reason as the batch pass: unlike a
|
||||
// missing workspace or a withdrawn grant, an open conversation is not a
|
||||
// condition that stops being true.
|
||||
const [lateLive] = rejectAlreadyLive([entry], liveSessionIds(), liveConversationIds()).skipped;
|
||||
if (lateLive) {
|
||||
failures.push(lateLive);
|
||||
unspent.delete(entry);
|
||||
continue;
|
||||
}
|
||||
// Capacity is re-checked per iteration, because this loop is itself
|
||||
// creating the sessions it counts. The offer can be a day old, so the
|
||||
// board may be fuller now than the plan assumed.
|
||||
const capMsg = sessionCapacityMessage(ctx.sessions, entry.owner);
|
||||
if (capMsg) {
|
||||
failures.push({ sessionId: entry.sessionId, reason: 'capacity-reached' });
|
||||
continue;
|
||||
}
|
||||
// A repo can be deleted between the boot that planned this and the click.
|
||||
if (!existsSync(entry.workingDir)) {
|
||||
failures.push({ sessionId: entry.sessionId, reason: 'workspace-missing' });
|
||||
continue;
|
||||
}
|
||||
// Multi-user workspace separation: the create route confines a non-admin's
|
||||
// workingDir to their own case space, and a grant can be withdrawn between
|
||||
// the session's creation and this restore, so the confinement is re-run
|
||||
// rather than inherited from the record. Keyed on the OWNER, not on the
|
||||
// caller: an admin spending another user's entry must be held to that
|
||||
// user's confinement, and `isWorkingDirAllowed` would wave an admin
|
||||
// through. The same reason the two grant re-checks below read
|
||||
// `saved.owner`.
|
||||
if (!(await isWorkingDirAllowedForUsername(entry.owner, entry.workingDir))) {
|
||||
// Left on offer: a withdrawn grant can be restored, unlike an already-open
|
||||
// conversation, so this is not the permanent kind of refusal.
|
||||
failures.push({ sessionId: entry.sessionId, reason: 'workspace-forbidden' });
|
||||
continue;
|
||||
}
|
||||
try {
|
||||
const saved = entry.state;
|
||||
const claudeModeConfig = await ctx.getClaudeModeConfig();
|
||||
const session = new Session({
|
||||
// The old id is reused on purpose: a pinned record, subagent parents,
|
||||
// window states and the lifecycle log all key off it, and the unpinned
|
||||
// record is gone, so there is nothing to collide with.
|
||||
id: saved.id,
|
||||
workingDir: saved.workingDir,
|
||||
mode: saved.mode,
|
||||
name: saved.name,
|
||||
// Without this the constructor re-infers ownership from the name, so a
|
||||
// session the user renamed by hand to something shaped like `w<n>-<case>`
|
||||
// comes back as `placeholder` and auto-naming overwrites their name on
|
||||
// the next prompt. The route persists below, so the loss would go to
|
||||
// disk. `restoreMuxSessions()` passes it for the same reason.
|
||||
nameSource: saved.nameSource,
|
||||
createdAt: saved.createdAt,
|
||||
mux: ctx.mux,
|
||||
useMux: true,
|
||||
// No `muxSession`: the reboot took the pane with it, so `startInteractive()`
|
||||
// takes its create branch and makes a fresh one.
|
||||
claudeMode: await resolveClaudeModeForUsername(claudeModeConfig.claudeMode, saved.owner),
|
||||
allowedTools: claudeModeConfig.allowedTools,
|
||||
resumeSessionId: entry.resumeConversationId,
|
||||
// Re-resolved against the owner's CURRENT grant, never replayed from the
|
||||
// record: a grant held when the record was written may be gone now.
|
||||
envOverrides: await clampEnvOverridesForOwner(
|
||||
saved.owner,
|
||||
(saved as { __envOverrides?: Record<string, string> }).__envOverrides
|
||||
),
|
||||
effort: saved.effort,
|
||||
attachmentHistory:
|
||||
(saved as { __attachmentHistory?: SessionAttachmentHistoryItem[] }).__attachmentHistory ??
|
||||
saved.attachmentHistory,
|
||||
lastSubmitAt: saved.lastSubmitAt,
|
||||
claudeSessionChain: saved.claudeSessionChain,
|
||||
lastActivityAt: saved.lastActivityAt,
|
||||
owner: saved.owner,
|
||||
parentSessionId: saved.parentSessionId,
|
||||
});
|
||||
|
||||
await ctx.addSession(session);
|
||||
// Before the listeners, because setupSessionListeners() reads the
|
||||
// image-watcher flag this phase restores; before the spawn, because the
|
||||
// custom-model environment and the nice priority shape the process.
|
||||
await ctx.reapplyPersistedSessionState(session, saved, 'before-spawn');
|
||||
await ctx.setupSessionListeners(session);
|
||||
await session.startInteractive();
|
||||
// The session's own history, applied only once the pane exists: on a
|
||||
// failed start these totals would belong to a session that never ran.
|
||||
// Both halves precede the route's OWN persist, which matters because a
|
||||
// constructed session carries none of this and `toState()` is written
|
||||
// wholesale, so persisting first would replace the fuller record with
|
||||
// the reduced one and drop the pin that keeps it from being pruned. A
|
||||
// listener-driven persist can still land inside the debounce window
|
||||
// while the pane starts; the write below repairs the record.
|
||||
// `rearmAutoResumeSchedule: false`: the saved stamp predates the reboot and
|
||||
// the pane is new, so honouring it would have every restored session type
|
||||
// `continue` into itself about a minute after one click. Auto-resume stays
|
||||
// enabled and re-arms on the next real limit message. This is also what the
|
||||
// module header promises ("comes back attached, idle and disarmed").
|
||||
await ctx.reapplyPersistedSessionState(session, saved, 'after-spawn', {
|
||||
rearmAutoResumeSchedule: false,
|
||||
});
|
||||
ctx.persistSessionState(session);
|
||||
|
||||
// A session without its workspace hooks goes silently blind: no stop or
|
||||
// idle events for respawn, no Approvals Inbox item, no red tab on a
|
||||
// blocking dialog. The boot-time sweep finished hours ago, so the click
|
||||
// path installs them itself. `hooks: 'always'` is the capability that says
|
||||
// this CLI installs Codeman's hooks into the workspace.
|
||||
if (workspaceHooksEnabled && getCli(session.mode)?.capabilities.hooks === 'always') {
|
||||
await applyWorkspaceHooks(session.workingDir, true).catch((err: unknown) =>
|
||||
console.warn(`[reboot-restore] hook install failed for ${session.workingDir}: ${getErrorMessage(err)}`)
|
||||
);
|
||||
}
|
||||
|
||||
// Both create paths seed this; without it a restored claude session's agent
|
||||
// skill falls back to writing out the whole ~150-line §0 preamble. Remote and
|
||||
// docker sessions never reach here (the plan rejects them as
|
||||
// `remote-or-docker`), so the local-only condition is structural.
|
||||
if (getCli(session.mode)?.capabilities.agentSkillInjection && (await ctx.getAgentSkillEnabled())) {
|
||||
await seedAgentSessionPreamble(session.id).catch((err: unknown) =>
|
||||
console.warn(`[agent-skill] preamble seed failed for ${session.id}: ${getErrorMessage(err)}`)
|
||||
);
|
||||
}
|
||||
|
||||
getLifecycleLog().log({ event: 'recovered', sessionId: session.id, name: session.name });
|
||||
// Every other open tab and phone needs this; the clicking tab already has
|
||||
// the response, and the client's handler is an idempotent upsert.
|
||||
ctx.broadcast(SseEvent.SessionCreated, ctx.getSessionStateWithRespawn(session));
|
||||
restored.push(toBannerItem(entry));
|
||||
} catch (err) {
|
||||
// One entry that will not start must not stop the rest of the pass, and
|
||||
// must not leave a registered session with no pane behind it: by this
|
||||
// point the session is in `ctx.sessions`, holds a tab-layout slot and has
|
||||
// listeners.
|
||||
//
|
||||
// Reaching this is rarer than it looks, measured against a real server:
|
||||
// the CLI resolver finds its binary by absolute path rather than through
|
||||
// PATH, and tmux falls back to another directory rather than failing when
|
||||
// it cannot enter the workspace, so neither of the two obvious "freshly
|
||||
// booted machine" failures throws. What is left is the mux layer itself
|
||||
// failing, which is why this path is defended rather than expected.
|
||||
console.error(`[reboot-restore] failed to rebuild ${entry.sessionId}:`, err);
|
||||
// Not cleanupSession(): that is the user-initiated delete, and it would
|
||||
// count this session's historical tokens into the lifetime totals, demote
|
||||
// a pinned record to `stopped` (which this pass reads as an intentional
|
||||
// kill, making the session permanently unrestorable) and delete the
|
||||
// workspace's `.claude-images`. This undoes only the construction.
|
||||
await ctx
|
||||
.discardPartiallyBuiltSession(entry.sessionId)
|
||||
.catch((discardErr: unknown) =>
|
||||
console.error(`[reboot-restore] discarding a failed rebuild failed: ${getErrorMessage(discardErr)}`)
|
||||
);
|
||||
failures.push({ sessionId: entry.sessionId, reason: 'rebuild-failed' });
|
||||
// Left on offer: the user can put the binary back and click again.
|
||||
continue;
|
||||
}
|
||||
unspent.delete(entry);
|
||||
}
|
||||
|
||||
if (restored.length > 0) {
|
||||
// A reboot leaves recovery with nothing alive to find, so its own block never
|
||||
// started the stats collector. This clears and re-arms its interval, so it is
|
||||
// safe to call whether or not the collector is already running.
|
||||
ctx.mux.startStatsCollection(STATS_COLLECTION_INTERVAL_MS);
|
||||
}
|
||||
|
||||
return { restored, skipped: failures };
|
||||
} finally {
|
||||
// Anything that never became a pane goes back on offer, including after a
|
||||
// throw, so a transient failure costs a retry rather than the whole plan.
|
||||
// Ends the flight: entries still parked for it come back if they are in
|
||||
// `unspent`, and a Dismiss that unparked them meanwhile wins.
|
||||
rebootRestoreRegistry.releaseFlight(owner, [...unspent]);
|
||||
rebootRestoreRegistry.endSpending(owner);
|
||||
}
|
||||
});
|
||||
|
||||
// ========== Drop it ==========
|
||||
|
||||
app.post('/api/reboot-restore/dismiss', async (req) => {
|
||||
const dismissed = rebootRestoreRegistry.clear(accessorFor(req));
|
||||
return { dismissed };
|
||||
});
|
||||
}
|
||||
@@ -11,7 +11,7 @@ import { homedir } from 'node:os';
|
||||
import { existsSync, statSync, mkdirSync, writeFileSync } from 'node:fs';
|
||||
import { execFile } from 'node:child_process';
|
||||
import fs from 'node:fs/promises';
|
||||
import { randomBytes } from 'node:crypto';
|
||||
import { randomBytes, randomUUID } from 'node:crypto';
|
||||
import { performance } from 'node:perf_hooks';
|
||||
import {
|
||||
ApiErrorCode,
|
||||
@@ -28,8 +28,10 @@ import {
|
||||
type GrokConfig,
|
||||
type DeepSeekConfig,
|
||||
type OmpConfig,
|
||||
type RemoteHost,
|
||||
} from '../../types.js';
|
||||
import { Session, isAltScreenStripMode, isExternalCliMode, isMuxAltScreenOnlyStripMode } from '../../session.js';
|
||||
import type { PaneCaptureOptions } from '../../mux-interface.js';
|
||||
import { SseEvent } from '../sse-events.js';
|
||||
import { webviewCapabilities } from '../../webview-capabilities.js';
|
||||
import {
|
||||
@@ -55,6 +57,12 @@ import {
|
||||
} from '../schemas.js';
|
||||
import { readCustomModelHosts } from '../../custom-model-hosts.js';
|
||||
import { applyCustomModelInjection, removeConfigDir } from '../../custom-model-injection-apply.js';
|
||||
import {
|
||||
getLlamaSwapStatus,
|
||||
triggerLlamaSwapLoad,
|
||||
exceedsSafeContextFloor,
|
||||
CLAUDE_MIN_SAFE_CONTEXT_TOKENS,
|
||||
} from './custom-model-routes.js';
|
||||
import { matchesPattern } from '../../config/cli-registry/patterns.js';
|
||||
import { ownerLayoutKey } from '../../tab-layout-persistence.js';
|
||||
import { TabLayoutValidationError } from '../../tab-layout.js';
|
||||
@@ -67,6 +75,13 @@ import {
|
||||
type WaitSignal,
|
||||
type SignalWaitResult,
|
||||
} from '../session-wait-registry.js';
|
||||
import {
|
||||
RemoteWakeRegistry,
|
||||
REMOTE_WAKE_REQUEST_READY_TIMEOUT_MS,
|
||||
createDefaultRemoteWakeDeps,
|
||||
isProbeable,
|
||||
type WakeableRemote,
|
||||
} from '../../remote-wake.js';
|
||||
import { clampWaitMs, MAX_BUFFER_SCAN_BYTES } from '../../config/agent-wait.js';
|
||||
import {
|
||||
autoConfigureRalph,
|
||||
@@ -89,6 +104,7 @@ import {
|
||||
} from '../route-helpers.js';
|
||||
import { buildAgentCaseMarker, writeAgentCaseMarker } from '../../agent-case-marker.js';
|
||||
import { canUsernameRunPrivilegedCommands, resolveClaudeModeForUsername } from '../../user-store.js';
|
||||
import { clampEnvOverridesForOwner } from '../../session-env-clamp.js';
|
||||
import { enabledClis, getCli } from '../../config/cli-registry/registry.js';
|
||||
import { resolveCliLaunchError } from '../../utils/cli-launcher.js';
|
||||
import { legacyConfigForMode } from '../../session-cli-registry-bridge.js';
|
||||
@@ -133,6 +149,7 @@ import {
|
||||
checkRemoteTmuxAvailable,
|
||||
readRemoteCases,
|
||||
readRemoteHosts,
|
||||
rehydrateRemoteHostFields,
|
||||
toAttachedSessionRemote,
|
||||
toSessionRemote,
|
||||
} from '../../remote-hosts.js';
|
||||
@@ -442,72 +459,6 @@ export async function _clampExternalCliBypassForOwner(
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Env-var keys a non-granted owner must not be able to set, because each one
|
||||
* hands back privilege the config clamp above just removed, or redirects a
|
||||
* credential-resolution endpoint.
|
||||
*
|
||||
* The DeepSeek three are reachable because `DSH_*` and `DEEPSEEK_*` are
|
||||
* allowlisted `envOverrides` prefixes (schemas.ts) — which they have to be, since
|
||||
* that is also how a user configures the harness's non-privileged knobs.
|
||||
*
|
||||
* - `DSH_PERMISSION_MODE` IS the harness's permission switch. Every other CLI's
|
||||
* bypass is a command-line FLAG, reachable only through the per-CLI config the
|
||||
* clamp already owns; this one is an env var, so the config clamp alone is
|
||||
* half a gate.
|
||||
* - `DSH_HOME` points the launcher at a profile tree, and a profile's plugin code
|
||||
* executes at BOOT, before any approval row can apply. A user who can write a
|
||||
* workspace can put a profile in it, so this is the wider of the two.
|
||||
* - `DEEPSEEK_BASE_URL` aims the provider endpoint, and `_configureCliEnv()`
|
||||
* forwards the SERVER's own `DEEPSEEK_API_KEY` into every dsh pane before
|
||||
* `applyEnvOverrides()` runs — so a non-granted owner who could set the base
|
||||
* URL would have the operator's API key sent as a bearer credential to a host
|
||||
* of their choosing. (`DEEPSEEK_API_KEY` itself stays overridable: supplying
|
||||
* your OWN key removes privilege rather than granting it.)
|
||||
* - `OMP_AUTH_BROKER_URL`/`OMP_AUTH_BROKER_TOKEN` are where omp resolves
|
||||
* credentials from — the same shape as `DEEPSEEK_BASE_URL` above, reachable
|
||||
* because `OMP_*` is an allowlisted prefix. Unlike DeepSeek, Codeman does not
|
||||
* forward any operator-held key into an omp pane today (omp's provider
|
||||
* credentials live in `~/.omp` config files, not env vars), so there is no
|
||||
* known concrete exfiltration path yet — clamped defensively anyway, since a
|
||||
* non-granted owner redirecting where a shared multi-tenant deployment
|
||||
* resolves auth from is not something to allow silently (found in
|
||||
* Ark0N/Codeman#353 review; omp's own knobs are otherwise mostly `PI_*`,
|
||||
* already allowlisted for pi and not addressed here — see resolveOmpHome()).
|
||||
*/
|
||||
function ownerClampedEnvKeys(): string[] {
|
||||
return enabledClis().flatMap((entry) => entry.capabilities.privilegedEnvKeys);
|
||||
}
|
||||
|
||||
/**
|
||||
* Env-var half of the multi-user bypass clamp.
|
||||
*
|
||||
* `clampExternalCliBypassForOwner()` clamps the per-CLI CONFIG, and for every CLI
|
||||
* but DeepSeek that is the whole story. Here it is not: `applyEnvOverrides()` runs
|
||||
* AFTER `_configureCliEnv()` in tmux-manager, so an override sent on the SAME
|
||||
* request lands last and wins, and a non-granted owner could restore
|
||||
* `danger-full-access` on the very request the config clamp downgraded.
|
||||
*
|
||||
* Keys are DROPPED rather than rewritten: dropping falls through to what
|
||||
* `_configureCliEnv()` exports, which is the clamped config and the server's own
|
||||
* `DSH_HOME`, i.e. exactly the intended state. No-op in single-user mode and for a
|
||||
* granted owner, like every other clamp here
|
||||
* (`canUsernameRunPrivilegedCommands()` returns true when `!isMultiUserMode()`),
|
||||
* and it returns the caller's own object untouched when there is nothing to strip.
|
||||
*/
|
||||
async function clampEnvOverridesForOwner(
|
||||
owner: string | undefined,
|
||||
envOverrides: Record<string, string> | undefined
|
||||
): Promise<Record<string, string> | undefined> {
|
||||
if (!envOverrides) return envOverrides;
|
||||
const keys = ownerClampedEnvKeys();
|
||||
if (!keys.some((key) => key in envOverrides)) return envOverrides;
|
||||
if (await canUsernameRunPrivilegedCommands(owner)) return envOverrides;
|
||||
const clamped = { ...envOverrides };
|
||||
for (const key of keys) delete clamped[key];
|
||||
return clamped;
|
||||
}
|
||||
|
||||
/** Test hook: the env-var half of the same multi-user safety gate. */
|
||||
export const _clampEnvOverridesForOwner = clampEnvOverridesForOwner;
|
||||
|
||||
@@ -814,10 +765,65 @@ export function resolveOmpConfigForCreate(
|
||||
return resolvedId ? { ...ompConfig, resumeSessionId: resolvedId } : ompConfig;
|
||||
}
|
||||
|
||||
/**
|
||||
* `RemoteHost` → the wake registry's host shape. They differ in one field name only
|
||||
* (`id` in host config vs `hostId` on a session's `remote`), but the rename is load-
|
||||
* bearing: the registry keys its per-host wake state on `hostId`. The proxy fields
|
||||
* travel too: they are what tells the registry its probe cannot reach this host.
|
||||
*/
|
||||
function wakeableHost(host: RemoteHost): WakeableRemote {
|
||||
return {
|
||||
hostId: host.id,
|
||||
label: host.label,
|
||||
host: host.host,
|
||||
port: host.port,
|
||||
wakeMac: host.wakeMac,
|
||||
wakeCommand: host.wakeCommand,
|
||||
jumpHost: host.jumpHost,
|
||||
socksProxy: host.socksProxy,
|
||||
extraSshOptions: host.extraSshOptions,
|
||||
};
|
||||
}
|
||||
|
||||
export function registerSessionRoutes(
|
||||
app: FastifyInstance,
|
||||
ctx: SessionPort & EventPort & ConfigPort & InfraPort & AuthPort & TabLayoutPort
|
||||
): void {
|
||||
ctx: SessionPort & EventPort & ConfigPort & InfraPort & AuthPort & TabLayoutPort,
|
||||
/** Test seam: inject a registry with fake IO instead of the real TCP/WoL probes. */
|
||||
options: { remoteWake?: RemoteWakeRegistry } = {}
|
||||
): RemoteWakeRegistry {
|
||||
// Wake-on-LAN for sleeping remote hosts (see remote-wake.ts). One registry per
|
||||
// route registration (= one web server) — the same shape as the process-wide
|
||||
// `sessionWaits` singleton, but without the global.
|
||||
//
|
||||
// ⚠️ The ONLY caller that may wake a host is the input route below. The
|
||||
// auto-reconnect watcher and boot recovery deliberately have no access to this
|
||||
// registry: waking there would re-wake the host seconds after every suspend, so
|
||||
// it could never stay asleep.
|
||||
const remoteWake =
|
||||
options.remoteWake ??
|
||||
new RemoteWakeRegistry(
|
||||
createDefaultRemoteWakeDeps({
|
||||
noteReconnected: (sessionId, success) => {
|
||||
// Duck-typed exactly like server.ts: TmuxManager owns the COD-108 backoff
|
||||
// state, and the port interface does not expose it.
|
||||
const mux = ctx.mux as unknown as { noteRemoteReconnect?: (id: string, ok: boolean) => void };
|
||||
mux.noteRemoteReconnect?.(sessionId, success);
|
||||
},
|
||||
broadcast: (event, payload) => ctx.broadcast(event, payload),
|
||||
log: (message) => console.log(message),
|
||||
// The session's `remote` block is a launch-time snapshot, so a wake target
|
||||
// configured later (banner's config dialog, or a hand-edited remote-hosts.json)
|
||||
// is resolved here — throttled by the registry, and the host config is
|
||||
// authoritative in BOTH directions (removing the field turns the feature off
|
||||
// for a live session too).
|
||||
resolveRemote: async (session) => {
|
||||
const remote = session.remote;
|
||||
if (!remote) return undefined;
|
||||
const hosts = await readRemoteHosts(CODEMAN_CONFIG_DIR);
|
||||
return rehydrateRemoteHostFields(remote, new Map(hosts.map((host) => [host.id, host])));
|
||||
},
|
||||
})
|
||||
);
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
// Auth
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
@@ -889,9 +895,33 @@ export function registerSessionRoutes(
|
||||
// creation (owned durable sessions) is handled by the dedicated case-create
|
||||
// endpoint below, which #145 consolidated remote-host resolution into.
|
||||
if (body.attachRemoteSession) {
|
||||
// Remote hosts are admin-only infrastructure everywhere else (the list answers
|
||||
// `[]` to a non-admin; write and discovery routes are `adminOnly`), and the wake
|
||||
// below spawns the host's `wakeCommand` or broadcasts a packet. So the gate comes
|
||||
// FIRST — before the host is even looked up — or an unprivileged account could
|
||||
// invoke that executable for any configured `hostId` and only then be told the
|
||||
// workingDir was outside its workspace (reproduced upstream: wake spy fired, 403).
|
||||
if (isMultiUserMode() && !isAdmin(req)) {
|
||||
return createErrorResponse(ApiErrorCode.FORBIDDEN, 'Remote hosts are admin-only in multi-user mode');
|
||||
}
|
||||
const { hostId, remoteSessionName } = body.attachRemoteSession;
|
||||
const host = (await readRemoteHosts(CODEMAN_CONFIG_DIR)).find((item) => item.id === hostId);
|
||||
if (!host) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Remote host not found');
|
||||
// An explicit wake request is the only thing that may wake a host, and the user
|
||||
// pressing Attach IS one (see quick-start for the same gate, and
|
||||
// `remote-wake.ts` for what must never call this). Without it a sleeping host
|
||||
// answers with an ssh failure that blames anything but the machine being asleep.
|
||||
const hostWake = await remoteWake.ensureHostAwake(wakeableHost(host), {
|
||||
timeoutMs: REMOTE_WAKE_REQUEST_READY_TIMEOUT_MS,
|
||||
// No session yet, so the wake events name their requester (multi-user routing).
|
||||
requestedBy: ownerFor(req),
|
||||
});
|
||||
if (hostWake === 'failed') {
|
||||
return createErrorResponse(
|
||||
ApiErrorCode.OPERATION_FAILED,
|
||||
`${host.label} did not come back after a wake-on-LAN request — nothing was attached`
|
||||
);
|
||||
}
|
||||
workingDir = `${host.username}@${host.host}:${remoteSessionName}`;
|
||||
remote = toAttachedSessionRemote(host, remoteSessionName, workingDir);
|
||||
}
|
||||
@@ -1208,13 +1238,83 @@ export function registerSessionRoutes(
|
||||
if (!endpoint) {
|
||||
return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
|
||||
}
|
||||
const contextLength = endpoint.modelContextLengths?.[body.modelId];
|
||||
|
||||
// Some CLIs (today: only claude) carry enough of their own fixed system-prompt/tool-
|
||||
// schema overhead that a small enough real context guarantees a first-message failure
|
||||
// no matter what CLAUDE_CODE_MAX_CONTEXT_TOKENS says — confirmed live at ~36.4K tokens
|
||||
// against a model configured with a real 16384-token context. Warn before committing
|
||||
// to a restart that's certain to fail, rather than letting the user discover it via a
|
||||
// cryptic 400 from the CLI itself. Answered by `confirmedContext` (or the legacy
|
||||
// `confirmed`, which still means both) — NOT by `confirmedSwap`: this warning is
|
||||
// about the caller's own session, and the swap warning below is about someone
|
||||
// else's, so an answer to one is not consent to the other.
|
||||
if (!(body.confirmed || body.confirmedContext) && exceedsSafeContextFloor(entry, contextLength)) {
|
||||
return {
|
||||
requiresContextWarning: true,
|
||||
modelId: body.modelId,
|
||||
contextLength,
|
||||
minSafeContextTokens: CLAUDE_MIN_SAFE_CONTEXT_TOKENS,
|
||||
};
|
||||
}
|
||||
|
||||
// llama.cpp runs exactly one model at a time; llama-swap unloads and reloads it on
|
||||
// demand, which can take anywhere from a few seconds to over a minute — long enough
|
||||
// that a session mid-swap looks indistinguishable from one that never left the native
|
||||
// backend. Feature-detected via llama-swap's own `GET /running` (a plain llama.cpp
|
||||
// server has no such endpoint and reads as `isLlamaSwap: false` — nothing to check).
|
||||
const swapStatus = await getLlamaSwapStatus(endpoint);
|
||||
const currentlyLoaded = swapStatus.running.find((r) => r.state === 'ready')?.model ?? swapStatus.running[0]?.model;
|
||||
// Distinct from targetReady below: this is ONLY about whether proceeding would evict a
|
||||
// model another session is actively using — true even if nothing is loaded at all yet
|
||||
// would be wrong here (nothing to evict), so this stays narrowly "a DIFFERENT model is
|
||||
// currently ready".
|
||||
const swapNeeded = swapStatus.isLlamaSwap && !!currentlyLoaded && currentlyLoaded !== body.modelId;
|
||||
// Whether the TARGET model itself is already the one loaded and ready — false whether
|
||||
// nothing is loaded yet, a different model is loaded, or this one is loaded but still
|
||||
// mid-load. Drives both the actual load trigger below and modelSwapInProgress in the
|
||||
// response; deliberately broader than swapNeeded, which only gates the confirmation ask.
|
||||
const targetReady = swapStatus.running.some((r) => r.model === body.modelId && r.state === 'ready');
|
||||
|
||||
// Only ask when switching would actually take the model away from another session
|
||||
// that is currently using it — never just because a swap is needed at all. Answered
|
||||
// by `confirmedSwap` (or the legacy `confirmed`). ⚠ It must NOT read
|
||||
// `confirmedContext`: this check runs second, and while the two shared one flag a
|
||||
// user who clicked past a too-small-context warning had already, silently, agreed to
|
||||
// evict another session's model.
|
||||
if (swapNeeded && !(body.confirmed || body.confirmedSwap)) {
|
||||
const conflicting = [...ctx.sessions.values()].filter(
|
||||
(s) =>
|
||||
s.id !== session.id && s.customModel?.endpointId === endpoint.id && s.customModel?.modelId === currentlyLoaded
|
||||
);
|
||||
if (conflicting.length > 0) {
|
||||
// Applying a custom model is ungated for any session owner, so in multi-user
|
||||
// mode a non-admin pointing their own session at a shared endpoint must not
|
||||
// learn another user's session names in the confirm dialog — with
|
||||
// autoNameSessions on, those names are that user's own prompts. The swap is
|
||||
// still blocked pending confirmation regardless of ownership (a foreign
|
||||
// session is just as real a disruption); only which ones get NAMED is scoped.
|
||||
const requestUser = getAuthUser(req);
|
||||
const affectedSessions = conflicting
|
||||
.filter((s) => canAccessOwned(requestUser, s.owner))
|
||||
.map((s) => ({ id: s.id, name: s.name }));
|
||||
return { requiresConfirmation: true, currentlyLoadedModel: currentlyLoaded, affectedSessions };
|
||||
}
|
||||
}
|
||||
|
||||
// A CLI whose config alone cannot select the model also gets its `model` launch param
|
||||
// forced (pi/omp `custom/<id>`, grok's block name). The argv engine DROPS a token that
|
||||
// fails its pattern rather than quoting it, which would silently launch the CLI on its
|
||||
// own default provider again, so refuse an id the pattern cannot carry up front.
|
||||
const modelSpec = entry.launch.params.model;
|
||||
const applied = applyCustomModelInjection(entry, endpoint, body.modelId, session.id);
|
||||
const applied = applyCustomModelInjection(
|
||||
entry,
|
||||
endpoint,
|
||||
body.modelId,
|
||||
session.id,
|
||||
contextLength,
|
||||
session.workingDir
|
||||
);
|
||||
if (!applied) {
|
||||
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${session.mode} has no known custom-model mechanism`);
|
||||
}
|
||||
@@ -1247,9 +1347,17 @@ export function registerSessionRoutes(
|
||||
removeConfigDir(previousConfigDir);
|
||||
}
|
||||
|
||||
// Actually kick off llama-swap's load now, rather than waiting on the restarted CLI's
|
||||
// own first prompt to do it — confirmed live that applying a selection alone never
|
||||
// reached the llama-swap server at all (nothing in its own logs), since llama-swap has
|
||||
// no "switch model" admin call, only a real inference request naming the model.
|
||||
if (swapStatus.isLlamaSwap && !targetReady) {
|
||||
triggerLlamaSwapLoad(endpoint, body.modelId);
|
||||
}
|
||||
|
||||
const restarted = await session.restartCli();
|
||||
persistAndBroadcastSession(ctx, session);
|
||||
return { customModel: session.customModel, restarted };
|
||||
return { customModel: session.customModel, restarted, modelSwapInProgress: swapStatus.isLlamaSwap && !targetReady };
|
||||
});
|
||||
|
||||
// ========== Delete Session ==========
|
||||
@@ -1279,6 +1387,8 @@ export function registerSessionRoutes(
|
||||
}
|
||||
|
||||
const session = findSessionOrFail(ctx, id, req);
|
||||
// Wake state is dropped by `cleanupSession` itself (server.ts), on EVERY cleanup
|
||||
// path — not here: the scheduled-run and admin paths clean up without this route.
|
||||
await ctx.cleanupSession(session.id, killMux, 'user_delete');
|
||||
return {};
|
||||
});
|
||||
@@ -1514,6 +1624,67 @@ export function registerSessionRoutes(
|
||||
// Terminal I/O (input, resize, buffer)
|
||||
// ═══════════════════════════════════════════════════════════════
|
||||
|
||||
// ========== Wake-on-LAN: state + manual trigger ==========
|
||||
//
|
||||
// Both routes are session-scoped (not host-scoped) because the wake flow needs the
|
||||
// SESSION: a woken host whose pane is not reattached is still a dead terminal, and an
|
||||
// exhausted COD-108 backoff never retries on its own. The probe in `/reachability` is
|
||||
// the same cheap TCP connect the input path uses and it NEVER wakes a host — the UI
|
||||
// decides that, with the button.
|
||||
|
||||
app.get('/api/sessions/:id/reachability', async (req) => {
|
||||
const { id } = req.params as { id: string };
|
||||
const session = findSessionOrFail(ctx, id, req);
|
||||
const remote = session.remote;
|
||||
if (!remote) {
|
||||
return { success: true, data: { reachable: true, probeable: true, wakeConfigured: 'none' as const } };
|
||||
}
|
||||
const force = (req.query as { force?: string })?.force === '1';
|
||||
// `reachable: null` + `probeable: false` for a host behind a jump host / SOCKS proxy:
|
||||
// the probe cannot reach it, so the UI shows no banner and stops polling.
|
||||
const reachable = await remoteWake.checkReachable(session, { force });
|
||||
return {
|
||||
success: true,
|
||||
data: {
|
||||
reachable,
|
||||
probeable: isProbeable(remote),
|
||||
wakeConfigured: await remoteWake.wakeConfigured(session),
|
||||
host: remote.host,
|
||||
label: remote.label,
|
||||
},
|
||||
};
|
||||
});
|
||||
|
||||
app.post('/api/sessions/:id/wake', async (req) => {
|
||||
const { id } = req.params as { id: string };
|
||||
const session = findSessionOrFail(ctx, id, req);
|
||||
if (!session.remote) {
|
||||
return createErrorResponse(ApiErrorCode.INVALID_INPUT, 'Not a remote session');
|
||||
}
|
||||
// The UI uses this to route to the host config dialog instead of a dead button.
|
||||
if (!(await remoteWake.hasWakeTarget(session))) {
|
||||
return createErrorResponse(
|
||||
ApiErrorCode.INVALID_INPUT,
|
||||
'No wake-on-LAN target configured for this host (set a MAC address or a wake command)'
|
||||
);
|
||||
}
|
||||
// The button is pressed from the SAME dashboard the create/attach paths are, under
|
||||
// the same reverse proxy — so it holds the request open the same way and needs the
|
||||
// same request budget, not the 90 s session default (see remote-wake.ts).
|
||||
const woke = await remoteWake.ensureAwake(session, {
|
||||
force: true,
|
||||
timeoutMs: REMOTE_WAKE_REQUEST_READY_TIMEOUT_MS,
|
||||
});
|
||||
return {
|
||||
success: true,
|
||||
data: {
|
||||
woke,
|
||||
reachable: await remoteWake.checkReachable(session),
|
||||
wakeConfigured: await remoteWake.wakeConfigured(session),
|
||||
},
|
||||
};
|
||||
});
|
||||
|
||||
// ========== Send Input ==========
|
||||
|
||||
app.post('/api/sessions/:id/input', async (req, reply) => {
|
||||
@@ -1555,6 +1726,42 @@ export function registerSessionRoutes(
|
||||
return {};
|
||||
}
|
||||
|
||||
// Wake-on-LAN (remote-wake.ts): a wake-enabled remote host that suspended leaves
|
||||
// the local ssh pane STALLED, and `send-keys` succeeds against it — the bytes
|
||||
// would vanish with no error anywhere. Give the registry the chance to probe the
|
||||
// host, wake it, reattach, and own delivery before we write into nothing.
|
||||
//
|
||||
// Costs nothing for non-wake hosts (the `wakeCommand` guard) or while the host is
|
||||
// known reachable inside the probe throttle window; the probe itself is a bare
|
||||
// TCP connect on wake-enabled hosts only, at most once per
|
||||
// REMOTE_WAKE_PROBE_MIN_INTERVAL_MS.
|
||||
if (!duplicate && (await remoteWake.hasWakeTarget(session))) {
|
||||
if (wantsWait) {
|
||||
// Send-and-wait keeps the response open anyway, so blocking on the wake is
|
||||
// simpler and more correct than buffering (buffering would break the wait).
|
||||
// A host that never comes back is an error here, as on the create/attach
|
||||
// paths: writing into the stalled pane would answer `delivered:true` plus a
|
||||
// timeout, which is the combination the API docs send callers to the wrong
|
||||
// recovery for.
|
||||
if (!(await remoteWake.ensureAwake(session))) {
|
||||
return createErrorResponse(
|
||||
ApiErrorCode.OPERATION_FAILED,
|
||||
`${session.remote?.label ?? 'the remote host'} did not come back after a wake-on-LAN request — nothing was sent`
|
||||
);
|
||||
}
|
||||
} else {
|
||||
const outcome = await remoteWake.handleInput(session, inputStr);
|
||||
// The registry holds the bytes and flushes them in order once the pane is
|
||||
// reattached. The client's ACK is this 200 — a tagged retry is deduped
|
||||
// (`shouldApplyInput` above already consumed the seq), so nothing is lost.
|
||||
// `buffered` is additive to the historical bare `{}`; `dropped` says the chunk
|
||||
// was over the wake buffer's cap and is GONE (a 200 with no field could not
|
||||
// tell delivered from buffered from dropped).
|
||||
if (outcome === 'buffered') return { buffered: true };
|
||||
if (outcome === 'dropped') return { buffered: true, dropped: true };
|
||||
}
|
||||
}
|
||||
|
||||
// Only a waiting request pays for the tmux probe: the browser's plain input path
|
||||
// (thousands of calls per session) must stay exec-free.
|
||||
const workerDead = wantsWait && workerIsDead(ctx.mux, session);
|
||||
@@ -1595,6 +1802,9 @@ export function registerSessionRoutes(
|
||||
|
||||
// Write input to PTY. Direct write is synchronous; writeViaMux
|
||||
// (tmux send-keys) is fire-and-forget to avoid blocking the HTTP response.
|
||||
// Every write here is `fromUser`: this route carries a person's prompt, or an
|
||||
// agent's on their behalf, so it may name the tab (Ralph, respawn, cron and
|
||||
// approvals write through the session directly and never say so).
|
||||
//
|
||||
// Because the response has already been sent by then, a failure there is the
|
||||
// one case the caller can never learn about — so the dedup bookkeeping is
|
||||
@@ -1617,32 +1827,32 @@ export function registerSessionRoutes(
|
||||
} else if (useMux && waitPromise) {
|
||||
// The response is already staying open for the wait, so the tmux write can be
|
||||
// awaited here. This is the ONE path where a writeViaMux failure is observable.
|
||||
const ok = await session.writeViaMux(inputStr).catch(() => false);
|
||||
const ok = await session.writeViaMux(inputStr, { fromUser: true }).catch(() => false);
|
||||
if (ok) {
|
||||
delivered = true;
|
||||
} else {
|
||||
console.warn(`[Server] writeViaMux failed for session ${id}, falling back to direct write`);
|
||||
delivered = session.write(inputStr);
|
||||
delivered = session.write(inputStr, { fromUser: true });
|
||||
if (!delivered) undoOnFailure();
|
||||
}
|
||||
} else if (useMux) {
|
||||
// Fire-and-forget: don't block the HTTP response on a tmux child process.
|
||||
// Fallback to a direct write on failure. Unchanged from before send-and-wait.
|
||||
session
|
||||
.writeViaMux(inputStr)
|
||||
.writeViaMux(inputStr, { fromUser: true })
|
||||
.then((ok) => {
|
||||
if (ok) return;
|
||||
console.warn(`[Server] writeViaMux failed for session ${id}, falling back to direct write`);
|
||||
if (!session.write(inputStr)) undoOnFailure();
|
||||
if (!session.write(inputStr, { fromUser: true })) undoOnFailure();
|
||||
})
|
||||
.catch(() => {
|
||||
if (!session.write(inputStr)) undoOnFailure();
|
||||
if (!session.write(inputStr, { fromUser: true })) undoOnFailure();
|
||||
});
|
||||
} else {
|
||||
// Same rollback. NOT an error response, deliberately: a session can
|
||||
// legitimately have no PTY yet (created but not started), and callers have
|
||||
// always been able to write to one without a 4xx.
|
||||
delivered = session.write(inputStr);
|
||||
delivered = session.write(inputStr, { fromUser: true });
|
||||
if (!delivered && tagged) {
|
||||
session.forgetInputSeq(clientId as string, seq as number);
|
||||
}
|
||||
@@ -1891,6 +2101,9 @@ export function registerSessionRoutes(
|
||||
console.error('[Server] send-key failed:', err);
|
||||
return createErrorResponse(ApiErrorCode.INTERNAL_ERROR, 'tmux send-keys failed');
|
||||
}
|
||||
// The bytes bypassed the session's write path, so tell the auto-name
|
||||
// tracker about them or the two lines of a prompt join with no separator.
|
||||
session.trackUserInput(hex.map((byte) => String.fromCharCode(parseInt(byte, 16))).join(''));
|
||||
return {};
|
||||
});
|
||||
|
||||
@@ -2691,14 +2904,16 @@ export function registerSessionRoutes(
|
||||
// returns null when unavailable, in which case we fall back to history.
|
||||
const muxName = session.muxName;
|
||||
const captureStartedAt = performance.now();
|
||||
// The visible path used to pass no options at all. It passes one now for a
|
||||
// single reason: `capturedGeometry` comes BACK on it, and the response has
|
||||
// to tell the client what size the frame it is about to render was built
|
||||
// for. See PaneCaptureOptions.capturedGeometry.
|
||||
const captureOpts: PaneCaptureOptions = isFullReload
|
||||
? { fullHistory: true, historyLimitLines: tmuxHistoryLimit, maxCaptureBytes: terminalBufferMaxBytes }
|
||||
: {};
|
||||
const liveMuxBuffer =
|
||||
muxName && typeof ctx.mux.captureActivePaneBuffer === 'function'
|
||||
? ctx.mux.captureActivePaneBuffer(
|
||||
muxName,
|
||||
isFullReload
|
||||
? { fullHistory: true, historyLimitLines: tmuxHistoryLimit, maxCaptureBytes: terminalBufferMaxBytes }
|
||||
: undefined
|
||||
)
|
||||
? ctx.mux.captureActivePaneBuffer(muxName, captureOpts)
|
||||
: null;
|
||||
const captureFinishedAt = performance.now();
|
||||
const hasLiveMuxBuffer = liveMuxBuffer !== null && liveMuxBuffer.length > 0;
|
||||
@@ -2844,6 +3059,25 @@ export function registerSessionRoutes(
|
||||
// what existed before the cut. The gap is what the indicator reports.
|
||||
retainedBytes: cleanBuffer.length,
|
||||
source,
|
||||
// The pane geometry this frame was drawn for. A visible-frame capture
|
||||
// positions every row absolutely, so a client whose terminal has fewer
|
||||
// rows than this overwrites its last line with the overflow and loses
|
||||
// the rows underneath. The client compares these against its own size.
|
||||
//
|
||||
// BOTH FIELDS ARE ABSENT unless this response really carries a capture,
|
||||
// and that is the honest answer rather than a gap to paper over. Two
|
||||
// separate things can leave a frame unpositioned. The cursor query is
|
||||
// what produces the absolute addressing in the first place, so a capture
|
||||
// that lost it returned a raw frame with no row positioning in it. And a
|
||||
// capture can report geometry and STILL hand back nothing: the
|
||||
// full-history path returns '' for a pane holding nothing visible, which
|
||||
// drops `source` to `history` while `capturedGeometry` is already
|
||||
// written, so the geometry has to be suppressed HERE rather than trusted
|
||||
// to be missing. Naming a size for a body that is the byte stream would
|
||||
// describe a frame that was never drawn and invite the client to repair
|
||||
// damage that does not exist.
|
||||
captureCols: hasLiveMuxBuffer ? captureOpts.capturedGeometry?.cols : undefined,
|
||||
captureRows: hasLiveMuxBuffer ? captureOpts.capturedGeometry?.rows : undefined,
|
||||
};
|
||||
});
|
||||
|
||||
@@ -3102,6 +3336,7 @@ export function registerSessionRoutes(
|
||||
effort,
|
||||
parentSessionId,
|
||||
agentOrigin,
|
||||
customModel,
|
||||
} = parseBody(QuickStartSchema, req.body);
|
||||
|
||||
// Resolved ONCE here: the same value labels a case directory this request creates
|
||||
@@ -3154,11 +3389,31 @@ export function registerSessionRoutes(
|
||||
grokConfig ||
|
||||
deepSeekConfig ||
|
||||
ompConfig ||
|
||||
openCodeConfig
|
||||
openCodeConfig ||
|
||||
customModel
|
||||
) {
|
||||
return createErrorResponse(
|
||||
ApiErrorCode.INVALID_INPUT,
|
||||
'envOverrides, effort, modelOverride, and per-CLI config are not supported for remote cases (they do not cross ssh). Configure the remote command via the host command override instead.'
|
||||
'envOverrides, effort, modelOverride, per-CLI config, and custom model endpoints are not supported for remote cases (they do not cross ssh). Configure the remote command via the host command override instead.'
|
||||
);
|
||||
}
|
||||
|
||||
// The user pressing "Run" on a case whose host is asleep IS an explicit wake
|
||||
// request (docs/remote-sessions.md §Wake-on-LAN), and the tmux probe below would
|
||||
// otherwise fail with "could not verify tmux on remote host …" — an ssh failure
|
||||
// that blames tmux for a machine that is merely suspended. Wired HERE, in the HTTP
|
||||
// route, and deliberately NOT in the shared session service: `cron-service.ts`
|
||||
// builds sessions through the service, and a wake down there would re-wake the
|
||||
// host on every schedule (the failure invariant #1 exists to prevent).
|
||||
const hostWake = await remoteWake.ensureHostAwake(wakeableHost(host), {
|
||||
timeoutMs: REMOTE_WAKE_REQUEST_READY_TIMEOUT_MS,
|
||||
// No session yet, so the wake events name their requester (multi-user routing).
|
||||
requestedBy: ownerFor(req),
|
||||
});
|
||||
if (hostWake === 'failed') {
|
||||
return createErrorResponse(
|
||||
ApiErrorCode.OPERATION_FAILED,
|
||||
`${host.label} did not come back after a wake-on-LAN request — the session was not started`
|
||||
);
|
||||
}
|
||||
|
||||
@@ -3167,6 +3422,20 @@ export function registerSessionRoutes(
|
||||
// surfaces a clear, structured error instead of a dead "tmux: command not found" pane.
|
||||
const tmuxCheck = await checkRemoteTmuxAvailable(host);
|
||||
if (!tmuxCheck.ok) {
|
||||
// An unreachable host and a host without tmux fail the same way over ssh, so the
|
||||
// probe's own message would send the user hunting for a tmux install. Ask the
|
||||
// registry (which just probed, when it woke the host) which of the two it is.
|
||||
// `=== false` on purpose: a proxied host answers `null` (the probe cannot reach
|
||||
// it), and an unknown verdict must not replace the real ssh error with
|
||||
// "not reachable" over a host that is fine.
|
||||
if ((await remoteWake.checkHostReachable(wakeableHost(host))) === false) {
|
||||
return createErrorResponse(
|
||||
ApiErrorCode.OPERATION_FAILED,
|
||||
hostWake === 'no-target'
|
||||
? `${host.label} (${host.host}) is not reachable, and this host has no wake-on-LAN target — configure a MAC address or a wake command first`
|
||||
: `${host.label} (${host.host}) is not reachable`
|
||||
);
|
||||
}
|
||||
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, tmuxCheck.error || 'remote host is missing tmux');
|
||||
}
|
||||
|
||||
@@ -3189,11 +3458,12 @@ export function registerSessionRoutes(
|
||||
grokConfig ||
|
||||
deepSeekConfig ||
|
||||
ompConfig ||
|
||||
openCodeConfig
|
||||
openCodeConfig ||
|
||||
customModel
|
||||
) {
|
||||
return createErrorResponse(
|
||||
ApiErrorCode.INVALID_INPUT,
|
||||
'envOverrides, effort, and per-CLI config are not supported for docker cases (they do not cross into the container). Configure the container via the docker host command override instead.'
|
||||
'envOverrides, effort, per-CLI config, and custom model endpoints are not supported for docker cases (they do not cross into the container). Configure the container via the docker host command override instead.'
|
||||
);
|
||||
}
|
||||
|
||||
@@ -3481,7 +3751,153 @@ export function registerSessionRoutes(
|
||||
);
|
||||
const qsTerminalHistoryConfig = await ctx.getTerminalHistoryConfig();
|
||||
const qsGatedEnvOverrides = await clampEnvOverridesForOwner(owner, envOverrides);
|
||||
const session = new Session({
|
||||
const qsResolvedOmpConfig = resolveOmpConfigForCreate(mode, resolvedCasePath, ompConfig);
|
||||
|
||||
// Custom Model Endpoint Profiles, applied AT CREATE TIME (docs/custom-model-endpoints-plan.md)
|
||||
// rather than via the dedicated restart-in-place route (POST /api/sessions/:id/custom-
|
||||
// model, still what an ALREADY-RUNNING session uses to switch later): computing the
|
||||
// injection before the process exists and launching directly on it avoids the visible
|
||||
// native-boot-then-restart the restart-after-launch design otherwise shows on every
|
||||
// custom-model run — most jarring on a CLI like Codex whose TUI fully reinitializes.
|
||||
// Mirrors the dedicated route's own checks (llama-swap conflict, unsupported CLI,
|
||||
// unknown endpoint, a model id the CLI's argv pattern can't carry) rather than trusting
|
||||
// a lighter version of them, since this is the same server-side authority reached a
|
||||
// different way, not a separate, less-checked path.
|
||||
let qsCustomModelEnvOverrides = qsGatedEnvOverrides;
|
||||
// Only the INJECTED keys (never the caller's envOverrides merged in) — this is what
|
||||
// setCustomModel() bookkeeping must be given below. The Session constructor already
|
||||
// applies qsCustomModelEnvOverrides (the full merged set) directly; re-merging that
|
||||
// full set into setCustomModel() would put CLAUDE_CODE_EFFORT_LEVEL back after the
|
||||
// constructor stripped it (see setCustomModel()'s own doc comment in session.ts).
|
||||
let qsCustomModelAppliedEnvOverrides: Record<string, string> | undefined;
|
||||
let qsCustomModelLaunchModel: string | undefined;
|
||||
let qsCustomModelSessionId: string | undefined;
|
||||
let qsCustomModelSwapInProgress = false;
|
||||
let qsCustomModelBookkeeping:
|
||||
| {
|
||||
endpointId: string;
|
||||
modelId: string;
|
||||
label?: string;
|
||||
envKeys: string[];
|
||||
configDir?: string;
|
||||
launchModel?: string;
|
||||
}
|
||||
| undefined;
|
||||
if (customModel) {
|
||||
const cmEntry = getCli(mode);
|
||||
if (!cmEntry) return createErrorResponse(ApiErrorCode.INVALID_INPUT, `No CLI registry entry for mode ${mode}`);
|
||||
if (cmEntry.capabilities.customModelInjection.kind === 'unsupported') {
|
||||
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${mode} has no known custom-model mechanism`);
|
||||
}
|
||||
const cmHosts = await readCustomModelHosts(CODEMAN_CONFIG_DIR);
|
||||
const cmEndpoint = cmHosts.find((h) => h.id === customModel.endpointId);
|
||||
if (!cmEndpoint) return createErrorResponse(ApiErrorCode.NOT_FOUND, 'Model endpoint not found');
|
||||
const cmContextLength = cmEndpoint.modelContextLengths?.[customModel.modelId];
|
||||
|
||||
// See the dedicated route's own comment for the full reasoning: some CLIs' own fixed
|
||||
// overhead can exceed a small enough real context on the very first message,
|
||||
// regardless of contextLengthVar. Warn before creating a session that's certain to
|
||||
// fail immediately.
|
||||
// See the dedicated route above for why this reads `confirmedContext` and never
|
||||
// `confirmedSwap`.
|
||||
if (
|
||||
!(customModel.confirmed || customModel.confirmedContext) &&
|
||||
exceedsSafeContextFloor(cmEntry, cmContextLength)
|
||||
) {
|
||||
return {
|
||||
requiresContextWarning: true,
|
||||
modelId: customModel.modelId,
|
||||
contextLength: cmContextLength,
|
||||
minSafeContextTokens: CLAUDE_MIN_SAFE_CONTEXT_TOKENS,
|
||||
};
|
||||
}
|
||||
|
||||
// See the dedicated route's own comment for the full reasoning: llama.cpp runs one
|
||||
// model at a time, llama-swap swaps on demand, and switching away from what another
|
||||
// live session is actively using deserves a warning, not a silent switch. There is no
|
||||
// "self" to exclude from the affected-sessions scan here — this session doesn't exist
|
||||
// yet.
|
||||
const cmSwapStatus = await getLlamaSwapStatus(cmEndpoint);
|
||||
const cmCurrentlyLoaded =
|
||||
cmSwapStatus.running.find((r) => r.state === 'ready')?.model ?? cmSwapStatus.running[0]?.model;
|
||||
const cmSwapNeeded = cmSwapStatus.isLlamaSwap && !!cmCurrentlyLoaded && cmCurrentlyLoaded !== customModel.modelId;
|
||||
// Broader than cmSwapNeeded (which only gates the confirmation ask above): true
|
||||
// whenever the TARGET model isn't already loaded and ready, including when nothing
|
||||
// is loaded at all yet. Drives the actual load trigger below.
|
||||
const cmTargetReady = cmSwapStatus.running.some((r) => r.model === customModel.modelId && r.state === 'ready');
|
||||
qsCustomModelSwapInProgress = cmSwapStatus.isLlamaSwap && !cmTargetReady;
|
||||
if (cmSwapNeeded && !(customModel.confirmed || customModel.confirmedSwap)) {
|
||||
const cmConflicting = [...ctx.sessions.values()].filter(
|
||||
(s) => s.customModel?.endpointId === cmEndpoint.id && s.customModel?.modelId === cmCurrentlyLoaded
|
||||
);
|
||||
if (cmConflicting.length > 0) {
|
||||
// Same reasoning as the dedicated /custom-model route above: the swap is
|
||||
// still blocked pending confirmation regardless of ownership, but a
|
||||
// non-admin caller only learns the names of sessions they can access.
|
||||
const cmRequestUser = getAuthUser(req);
|
||||
const cmAffectedSessions = cmConflicting
|
||||
.filter((s) => canAccessOwned(cmRequestUser, s.owner))
|
||||
.map((s) => ({ id: s.id, name: s.name }));
|
||||
return {
|
||||
requiresConfirmation: true,
|
||||
currentlyLoadedModel: cmCurrentlyLoaded,
|
||||
affectedSessions: cmAffectedSessions,
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Minted ourselves (rather than left to Session's own default) so the injection
|
||||
// below — and any configDir it writes — can target the REAL id the session launches
|
||||
// with, not a placeholder: `new Session({ id: ... })` accepts an explicit id for
|
||||
// exactly this reason.
|
||||
qsCustomModelSessionId = randomUUID();
|
||||
const cmApplied = applyCustomModelInjection(
|
||||
cmEntry,
|
||||
cmEndpoint,
|
||||
customModel.modelId,
|
||||
qsCustomModelSessionId,
|
||||
cmContextLength,
|
||||
resolvedCasePath
|
||||
);
|
||||
if (!cmApplied) {
|
||||
return createErrorResponse(ApiErrorCode.OPERATION_FAILED, `${mode} has no known custom-model mechanism`);
|
||||
}
|
||||
const cmModelSpec = cmEntry.launch.params.model;
|
||||
if (
|
||||
cmApplied.launchModel !== undefined &&
|
||||
cmModelSpec?.type === 'token' &&
|
||||
!matchesPattern(cmModelSpec.pattern, cmApplied.launchModel)
|
||||
) {
|
||||
removeConfigDir(cmApplied.configDir);
|
||||
return createErrorResponse(
|
||||
ApiErrorCode.INVALID_INPUT,
|
||||
`Model id ${JSON.stringify(customModel.modelId)} cannot be passed to ${mode} on its command line`
|
||||
);
|
||||
}
|
||||
|
||||
qsCustomModelEnvOverrides = { ...qsGatedEnvOverrides, ...cmApplied.envOverrides };
|
||||
qsCustomModelAppliedEnvOverrides = cmApplied.envOverrides;
|
||||
qsCustomModelLaunchModel = cmApplied.launchModel;
|
||||
qsCustomModelBookkeeping = {
|
||||
endpointId: cmEndpoint.id,
|
||||
modelId: customModel.modelId,
|
||||
label: cmEndpoint.label,
|
||||
envKeys: cmApplied.envKeys,
|
||||
configDir: cmApplied.configDir,
|
||||
launchModel: cmApplied.launchModel,
|
||||
};
|
||||
|
||||
// Actually kick off llama-swap's load now — see the dedicated apply route's own
|
||||
// comment on triggerLlamaSwapLoad for why this can't just wait on the launched CLI's
|
||||
// first prompt. Fired here, before the session is even created, so the load starts
|
||||
// concurrently with Claude/Codex/etc. booting rather than after.
|
||||
if (qsCustomModelSwapInProgress) {
|
||||
triggerLlamaSwapLoad(cmEndpoint, customModel.modelId);
|
||||
}
|
||||
}
|
||||
|
||||
const qsSessionOptions: ConstructorParameters<typeof Session>[0] = {
|
||||
id: qsCustomModelSessionId,
|
||||
workingDir: resolvedCasePath,
|
||||
name: sessionName ? sessionName.slice(0, MAX_SESSION_NAME_LENGTH) : '',
|
||||
mux: ctx.mux,
|
||||
@@ -3499,15 +3915,42 @@ export function registerSessionRoutes(
|
||||
piConfig: mode === 'pi' ? qsGatedPiConfig : undefined,
|
||||
grokConfig: mode === 'grok' ? qsGatedGrokConfig : undefined,
|
||||
deepSeekConfig: mode === 'deepseek' ? qsGatedDeepSeekConfig : undefined,
|
||||
ompConfig: resolveOmpConfigForCreate(mode, resolvedCasePath, ompConfig),
|
||||
envOverrides: qsGatedEnvOverrides,
|
||||
ompConfig: qsResolvedOmpConfig,
|
||||
envOverrides: qsCustomModelEnvOverrides,
|
||||
effort,
|
||||
remote,
|
||||
docker,
|
||||
resumeSessionId: dockerResumeId,
|
||||
tmuxHistoryLimit: qsTerminalHistoryConfig.tmuxHistoryLimit,
|
||||
parentSessionId: qsParentSessionId,
|
||||
});
|
||||
};
|
||||
// Force the custom-model selection's launchModel (pi/omp `custom/<id>`, grok's
|
||||
// `[model.<name>]` block name) onto whichever config field the registry says the
|
||||
// CLI's `model` launch param lives in — mirrors Session._withCustomModelLaunchModel,
|
||||
// which the restart-in-place path already uses, rather than a hardcoded per-CLI
|
||||
// branch here that a CLI landing its injection recipe later would silently miss.
|
||||
if (qsCustomModelLaunchModel !== undefined) {
|
||||
const qsCustomModelField = getCli(mode)?.launch.legacyConfigField;
|
||||
if (qsCustomModelField) {
|
||||
const qsSessionOptionsBag = qsSessionOptions as unknown as Record<string, unknown>;
|
||||
qsSessionOptionsBag[qsCustomModelField] = {
|
||||
...((qsSessionOptionsBag[qsCustomModelField] as Record<string, unknown>) ?? {}),
|
||||
model: qsCustomModelLaunchModel,
|
||||
};
|
||||
} else {
|
||||
qsSessionOptions.model = qsCustomModelLaunchModel;
|
||||
}
|
||||
}
|
||||
const session = new Session(qsSessionOptions);
|
||||
|
||||
// Records the selection for session.customModel/getCustomModelForPersist() and future
|
||||
// clear/switch calls — the actual env vars and launch-model config are already part of
|
||||
// the launch above (constructor envOverrides, piConfig/grokConfig/ompConfig.model), so
|
||||
// this is bookkeeping only, never a restart: setCustomModel() is synchronous state, no
|
||||
// tmux IO of its own (see its own doc comment in session.ts).
|
||||
if (qsCustomModelBookkeeping) {
|
||||
session.setCustomModel(qsCustomModelBookkeeping, qsCustomModelAppliedEnvOverrides);
|
||||
}
|
||||
|
||||
// Auto-detect completion phrase from CLAUDE.md BEFORE broadcasting
|
||||
// so the initial state already has the phrase configured (only if globally enabled)
|
||||
@@ -3603,6 +4046,7 @@ export function registerSessionRoutes(
|
||||
sessionId: session.id,
|
||||
casePath: resolvedCasePath,
|
||||
caseName,
|
||||
...(customModel ? { modelSwapInProgress: qsCustomModelSwapInProgress } : {}),
|
||||
};
|
||||
} catch (err) {
|
||||
// Clean up session on error to prevent orphaned resources
|
||||
@@ -4725,4 +5169,10 @@ export function registerSessionRoutes(
|
||||
|
||||
return { path: filepath, filename };
|
||||
});
|
||||
|
||||
// Returned so the server can own the registry's LIFETIME (drop state when a session is
|
||||
// cleaned up on any of its paths, resolve in-flight wakes on shutdown). The wake-CAPABLE
|
||||
// code stays here: `test/remote-wake.test.ts` pins that `server.ts` calls nothing but
|
||||
// `drop`/`stop` on this handle, so no timer path can reach a wake through it.
|
||||
return remoteWake;
|
||||
}
|
||||
|
||||
@@ -185,7 +185,8 @@ export function registerWsRoutes(app: FastifyInstance, ctx: SessionPort, getHost
|
||||
// Typed input from a claim-holding desktop keeps the claim "hot"
|
||||
// and re-asserts the desktop layout after a mobile override.
|
||||
if (holdsDesktopClaim) session.noteDesktopActivity();
|
||||
delivered = session.write(msg.d);
|
||||
// Browser keystrokes are the user's own, so they may name the tab.
|
||||
delivered = session.write(msg.d, { fromUser: true });
|
||||
// A session whose PTY is gone swallows the write. ACKing anyway told
|
||||
// the client to drop the frame from its durable queue and left the seq
|
||||
// burnt, so the retry that reliable delivery exists for was rejected as
|
||||
|
||||
@@ -19,6 +19,7 @@ import {
|
||||
} from '../config/terminal-history.js';
|
||||
import { MAX_EDITABLE_BYTES } from '../config/file-editing.js';
|
||||
import { MIN_MATCH_LENGTH, MAX_MATCH_LENGTH } from '../config/agent-wait.js';
|
||||
import { MAX_WAKE_MACS } from '../config/remote-wake-limits.js';
|
||||
import { enabledCliIds, enabledClis } from '../config/cli-registry/registry.js';
|
||||
import type { SessionMode } from '../types.js';
|
||||
|
||||
@@ -737,6 +738,36 @@ export const RemoteHostSchema = z.object({
|
||||
.max(32)
|
||||
.optional(),
|
||||
commands: RemoteCommandOverridesSchema,
|
||||
// Wake-on-LAN: a single executable path (no arguments, no shell) run to power a
|
||||
// SLEEPING host back on, e.g. `/home/joe/bin/whuff`. Executed via spawn without
|
||||
// a shell, so there is no shell layer to escape; the regexes are belt-and-braces
|
||||
// (and the no-whitespace rule rejects an argument list before it can fail as a
|
||||
// confusing ENOENT at wake time). See docs/remote-sessions.md §Wake-on-LAN.
|
||||
wakeCommand: z
|
||||
.string()
|
||||
.min(1)
|
||||
.max(4096)
|
||||
.regex(/^\S+$/, 'Wake command must be a single executable path (no arguments)')
|
||||
.regex(NO_SHELL_META, 'Invalid characters in wake command')
|
||||
.optional(),
|
||||
// Wake-on-LAN MAC address(es), comma-separated. Structural: only hex pairs with
|
||||
// `:`/`-` separators, so nothing here can be a shell token even by accident (the
|
||||
// value never reaches a shell — Codeman builds the magic packet itself).
|
||||
wakeMac: z
|
||||
.string()
|
||||
.min(11)
|
||||
.max(128)
|
||||
.regex(
|
||||
/^[0-9a-fA-F]{2}([:-][0-9a-fA-F]{2}){5}(\s*,\s*[0-9a-fA-F]{2}([:-][0-9a-fA-F]{2}){5})*$/,
|
||||
'Wake MAC must be one or more MAC addresses, comma-separated'
|
||||
)
|
||||
// ⚠ The character cap admits seven MACs while parseMacList takes at most
|
||||
// MAX_WAKE_MACS, all-or-nothing. Without this the extra ones validated, persisted,
|
||||
// and then resolved to NO wake target, so the host read as unconfigured.
|
||||
.refine((value) => value.split(',').length <= MAX_WAKE_MACS, {
|
||||
message: `Wake MAC accepts at most ${MAX_WAKE_MACS} comma-separated addresses`,
|
||||
})
|
||||
.optional(),
|
||||
});
|
||||
|
||||
export const RemoteCaseLinkSchema = z.object({
|
||||
@@ -1033,6 +1064,27 @@ export const QuickStartSchema = z.object({
|
||||
* because it takes an existing `workingDir` and so never creates a directory to label.
|
||||
*/
|
||||
agentOrigin: z.string().max(64).optional(),
|
||||
/**
|
||||
* Custom Model Endpoint Profiles (docs/custom-model-endpoints-plan.md): launches directly
|
||||
* on this saved endpoint/model instead of the mode's native backend, computed server-side
|
||||
* from the admin-configured endpoint store the same way `POST /api/sessions/:id/custom-
|
||||
* model` does — never trusting raw env values from the client. One-shot, launch-time
|
||||
* equivalent of that route: no restart, so no visible relaunch (that route's restart-in-
|
||||
* place is still what an ALREADY-RUNNING session uses to switch later). Rejected for
|
||||
* remote/docker cases, same reasoning as `envOverrides` above. The three confirmation
|
||||
* flags mirror that route's fields; see `SessionCustomModelSchema` for why there are
|
||||
* two specific ones rather than the single legacy `confirmed`.
|
||||
*/
|
||||
customModel: z
|
||||
.object({
|
||||
endpointId: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid endpoint id'),
|
||||
modelId: z.string().min(1).max(200),
|
||||
confirmed: z.boolean().optional(),
|
||||
confirmedContext: z.boolean().optional(),
|
||||
confirmedSwap: z.boolean().optional(),
|
||||
})
|
||||
.strict()
|
||||
.optional(),
|
||||
});
|
||||
|
||||
// ========== Hook Events ==========
|
||||
@@ -1161,6 +1213,20 @@ const NotificationEventSchema = z
|
||||
})
|
||||
.optional();
|
||||
|
||||
/**
|
||||
* Body of `POST /api/reboot-restore/restore`.
|
||||
*
|
||||
* `sessionIds` restores a subset, and omitting it restores everything the caller
|
||||
* can see. The ids are session ids from `GET /api/reboot-restore`, and an id the
|
||||
* caller does not own is ignored rather than refused, matching how the session
|
||||
* list scopes rather than 403s.
|
||||
*/
|
||||
export const RebootRestoreRequestSchema = z
|
||||
.object({
|
||||
sessionIds: z.array(z.string().max(128)).max(200).optional(),
|
||||
})
|
||||
.strict();
|
||||
|
||||
export const SettingsUpdateSchema = z
|
||||
.object({
|
||||
// User-facing product branding. This changes browser/UI copy only; package,
|
||||
@@ -1230,6 +1296,13 @@ export const SettingsUpdateSchema = z
|
||||
* already pending immediately.
|
||||
*/
|
||||
approvalsInboxEnabled: z.boolean().optional(),
|
||||
/**
|
||||
* Auto-name sessions: a placeholder tab (`w3-case`) takes its first real
|
||||
* prompt as a title (`w3-case: fix the login redirect`). Synced, default
|
||||
* OFF: the prompt lands in mux-sessions.json, every session:updated
|
||||
* broadcast and /api/search, which is the user's choice to make.
|
||||
*/
|
||||
autoNameSessions: z.boolean().optional(),
|
||||
/**
|
||||
* Read My Mind (docs/readmymind-plan.md): capture the user's submitted
|
||||
* prompts into per-case intent profiles. SYNCED, default OFF (opt-in:
|
||||
@@ -1911,6 +1984,17 @@ export const CustomModelHostSchema = z.object({
|
||||
authStyle: z.enum(['bearer', 'api-key']).optional(),
|
||||
models: z.array(z.string().max(200)).max(200).optional(),
|
||||
lastDiscoveredAt: z.string().max(64).optional(),
|
||||
// The Run-menu picker's per-endpoint default; validated against `models` at the
|
||||
// route layer (schema-level cross-field checks can't see the array narrowed the
|
||||
// same way a `.refine()` closure could, and the route already re-reads the stored
|
||||
// host to apply it, so the check belongs there once, not duplicated into a refine
|
||||
// that would run on every unrelated field edit too).
|
||||
defaultModelId: z.string().max(200).optional(),
|
||||
// Server-populated by discovery (custom-model-routes.ts); accepted here only so a client
|
||||
// round-tripping the GET response back through PUT (edit-save) doesn't drop it.
|
||||
modelContextLengths: z.record(z.string().max(200), z.number().int().positive().max(100_000_000)).optional(),
|
||||
// Same reasoning as modelContextLengths above.
|
||||
modelSizesGB: z.record(z.string().max(200), z.number().positive().max(100_000)).optional(),
|
||||
});
|
||||
|
||||
/** POST /api/sessions/:id/custom-model — apply or clear a session's custom-model selection. */
|
||||
@@ -1918,6 +2002,22 @@ export const CustomModelSelectionSchema = z.union([
|
||||
z.object({
|
||||
endpointId: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid endpoint id'),
|
||||
modelId: z.string().min(1).max(200),
|
||||
/**
|
||||
* Two DIFFERENT questions can block a launch, and answering one is not consent to
|
||||
* the other: `confirmedContext` answers "this model's context window is below the
|
||||
* floor for this CLI", which affects only the caller, while `confirmedSwap` answers
|
||||
* "loading this will unload the model another session is using", which affects
|
||||
* someone else. They were one flag until the context check (which runs first)
|
||||
* silently spent the swap answer too, so a user clicking "launch anyway" past a
|
||||
* too-small context evicted another session's model without ever being asked.
|
||||
*
|
||||
* `confirmed` is the original single flag and still means BOTH, because it shipped
|
||||
* in the HTTP-API-only cut of this feature and an existing caller must keep working.
|
||||
* New callers should send the specific one they actually asked about.
|
||||
*/
|
||||
confirmed: z.boolean().optional(),
|
||||
confirmedContext: z.boolean().optional(),
|
||||
confirmedSwap: z.boolean().optional(),
|
||||
}),
|
||||
z.object({ clear: z.literal(true) }),
|
||||
]);
|
||||
|
||||
+408
-6
@@ -39,8 +39,12 @@ import { fileURLToPath } from 'node:url';
|
||||
import { existsSync, mkdirSync, readFileSync, chmodSync, rmSync, statSync } from 'node:fs';
|
||||
import fs from 'node:fs/promises';
|
||||
import { execSync } from 'node:child_process';
|
||||
import { hostname as getHostname } from 'node:os';
|
||||
import { hostname as getHostname, uptime as osUptime } from 'node:os';
|
||||
import { looksLikeHostReboot, newestPersistedActivity, planRebootRestore } from '../reboot-restore.js';
|
||||
import { rebootRestoreRegistry } from './reboot-restore-registry.js';
|
||||
import { dataPath, getDataDir, CODEMAN_INSTANCE } from '../config/instance.js';
|
||||
import { readRemoteHosts, rehydrateRemoteHostFields } from '../remote-hosts.js';
|
||||
import type { RemoteWakeRegistry } from '../remote-wake.js';
|
||||
import { normalizeBasePath, stripBasePath, joinBasePath } from '../config/base-path.js';
|
||||
import { GLYPH, palette } from '../cli-style.js';
|
||||
import { getHookSecret } from '../config/hook-secret.js';
|
||||
@@ -67,7 +71,7 @@ import {
|
||||
import { imageWatcher } from '../image-watcher.js';
|
||||
import { workflowRunWatcher, summarizeRun } from '../workflow-run-watcher.js';
|
||||
import { attachmentRegistry, buildFileThumbnailRoute, registerExternalAttachment } from '../attachment-registry.js';
|
||||
import { getCli } from '../config/cli-registry/registry.js';
|
||||
import { getCli, enabledClis } from '../config/cli-registry/registry.js';
|
||||
import { readCustomModelHosts } from '../custom-model-hosts.js';
|
||||
import { applyCustomModelInjection, customModelConfigDir, removeConfigDir } from '../custom-model-injection-apply.js';
|
||||
import type { CustomModelBookkeeping } from '../types/session.js';
|
||||
@@ -171,6 +175,7 @@ import {
|
||||
registerScheduledRoutes,
|
||||
registerHookEventRoutes,
|
||||
registerApprovalRoutes,
|
||||
registerRebootRestoreRoutes,
|
||||
registerReadMyMindRoutes,
|
||||
registerStatusTelemetryRoutes,
|
||||
registerSystemRoutes,
|
||||
@@ -190,6 +195,11 @@ import {
|
||||
registerWebviewRoutes,
|
||||
registerTabLayoutRoutes,
|
||||
registerCustomModelRoutes,
|
||||
refreshAllCustomModelHosts,
|
||||
readCustomModelEndpointsEnabled,
|
||||
closeAllLlamaSwapLogTails,
|
||||
detectCustomModelSwapDisplacements,
|
||||
pruneIdleLlamaSwapLogTails,
|
||||
tryWebviewRefererFallback,
|
||||
} from './routes/index.js';
|
||||
import { isLostWebviewFrameNavigation } from './webview-proxy.js';
|
||||
@@ -202,11 +212,32 @@ const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
// while capping growth of `sseClientsById` and blocking pathological inputs.
|
||||
const SSE_CLIENT_ID_RE = /^[A-Za-z0-9_-]{8,64}$/;
|
||||
const CODEX_USAGE_POLL_INTERVAL_MS = 5 * 60_000;
|
||||
const CUSTOM_MODEL_REDISCOVER_INTERVAL_MS = 5 * 60_000;
|
||||
// Much shorter than the model-LIST refresh above on purpose: this catches an actual
|
||||
// eviction (a session's model no longer loaded, silently swapped out by another
|
||||
// session's use), which the user wants to know about promptly, not once every 5
|
||||
// minutes. Cheap either way — one /running GET per distinct endpoint with at least
|
||||
// one live custom-model session, not per session.
|
||||
const CUSTOM_MODEL_SWAP_CHECK_INTERVAL_MS = 20_000;
|
||||
|
||||
function escapeHtmlText(value: string): string {
|
||||
return value.replaceAll('&', '&').replaceAll('<', '<').replaceAll('>', '>');
|
||||
}
|
||||
|
||||
/**
|
||||
* Escapes a JSON string for safe embedding as the body of an inline `<script>`
|
||||
* tag: `<` becomes the six-character sequence `<`, which both a JSON
|
||||
* parser and a plain JS string literal decode back to `<` (both treat
|
||||
* `\uXXXX` identically), but which can never itself form the two literal
|
||||
* characters `<` `/` a browser's HTML tokenizer looks for to end the tag. A
|
||||
* value containing a literal `</script>` would otherwise close the tag early
|
||||
* and turn the rest of the document into inert script-body text. Exported so
|
||||
* it unit-tests without constructing a WebServer (which needs a real tmux).
|
||||
*/
|
||||
export function escapeScriptJson(json: string): string {
|
||||
return json.replace(/</g, '\\u003c');
|
||||
}
|
||||
|
||||
import {
|
||||
SESSIONS_LIST_CACHE_TTL,
|
||||
SCHEDULED_CLEANUP_INTERVAL,
|
||||
@@ -265,8 +296,18 @@ export class WebServer extends EventEmitter {
|
||||
// Store session listener references for explicit cleanup (prevents memory leaks)
|
||||
private sessionListenerRefs: Map<string, SessionListenerRefs> = new Map();
|
||||
private scheduledRuns: Map<string, ScheduledRun> = new Map();
|
||||
/** De-dupe state for the swap-displacement sweep — see detectCustomModelSwapDisplacements. */
|
||||
private _customModelDisplacedNotified: Set<string> = new Set();
|
||||
/** Cron service (assigned in setupRoutes). */
|
||||
private cronService!: CronService;
|
||||
/**
|
||||
* Wake-on-LAN registry, returned by `registerSessionRoutes`. Held for its LIFETIME
|
||||
* only — `drop()` on session cleanup, `stop()` on shutdown. Waking from here would
|
||||
* re-wake a host on every timer tick (the invariant `remote-wake.ts` documents), so
|
||||
* the wiring guard in `test/remote-wake.test.ts` pins that this file calls nothing
|
||||
* but `drop`/`stop` on it.
|
||||
*/
|
||||
private remoteWake: RemoteWakeRegistry | null = null;
|
||||
private sse: SseStreamManager;
|
||||
private store = getStore();
|
||||
private tabLayouts!: TabLayoutService;
|
||||
@@ -668,6 +709,8 @@ export class WebServer extends EventEmitter {
|
||||
setupSessionListeners: this.setupSessionListeners.bind(this),
|
||||
persistSessionState: this.persistSessionState.bind(this),
|
||||
persistSessionStateNow: this._persistSessionStateNow.bind(this),
|
||||
reapplyPersistedSessionState: this.reapplyPersistedSessionState.bind(this),
|
||||
discardPartiallyBuiltSession: this.discardPartiallyBuiltSession.bind(this),
|
||||
getSessionStateWithRespawn: this.getSessionStateWithRespawn.bind(this),
|
||||
// EventPort
|
||||
broadcast: this.broadcast.bind(this),
|
||||
@@ -1060,11 +1103,14 @@ export class WebServer extends EventEmitter {
|
||||
registerScheduledRoutes(this.app, ctx);
|
||||
registerHookEventRoutes(this.app, ctx);
|
||||
registerApprovalRoutes(this.app, ctx);
|
||||
registerRebootRestoreRoutes(this.app, ctx);
|
||||
registerReadMyMindRoutes(this.app, ctx);
|
||||
registerStatusTelemetryRoutes(this.app, ctx);
|
||||
registerSystemRoutes(this.app, ctx);
|
||||
registerCaseRoutes(this.app, ctx);
|
||||
registerSessionRoutes(this.app, ctx);
|
||||
// The registry's lifetime is the server's: it drops per-session wake state on every
|
||||
// cleanup path and resolves in-flight wakes on shutdown.
|
||||
this.remoteWake = registerSessionRoutes(this.app, ctx);
|
||||
registerRespawnRoutes(this.app, ctx);
|
||||
registerRalphRoutes(this.app, ctx);
|
||||
registerPlanRoutes(this.app, ctx);
|
||||
@@ -1206,7 +1252,13 @@ export class WebServer extends EventEmitter {
|
||||
return undefined;
|
||||
}
|
||||
try {
|
||||
return applyCustomModelInjection(entry, endpoint, saved.modelId, session.id)?.envOverrides;
|
||||
return applyCustomModelInjection(
|
||||
entry,
|
||||
endpoint,
|
||||
saved.modelId,
|
||||
session.id,
|
||||
endpoint.modelContextLengths?.[saved.modelId]
|
||||
)?.envOverrides;
|
||||
} catch (err) {
|
||||
console.warn('[WebServer] Failed to rebuild custom-model env on recovery:', err);
|
||||
return undefined;
|
||||
@@ -1301,6 +1353,10 @@ export class WebServer extends EventEmitter {
|
||||
session.ralphTracker.stopWatchingFixPlan();
|
||||
}
|
||||
|
||||
// Custom Model Endpoint Profiles: drop this session's swap-displacement notify flag
|
||||
// (see _checkCustomModelSwapDisplacements below) so it can't linger in that Set forever.
|
||||
this._customModelDisplacedNotified.delete(sessionId);
|
||||
|
||||
// Kill all subagents spawned by this session (scoped to sessionId to avoid cross-session kills)
|
||||
if (session && killMux) {
|
||||
try {
|
||||
@@ -1455,6 +1511,11 @@ export class WebServer extends EventEmitter {
|
||||
sessionWaits.notifySignal(sessionId, 'exit');
|
||||
sessionWaits.cancelAll(sessionId);
|
||||
approvalInbox.resolveForSession(sessionId, 'session_ended');
|
||||
// Wake state goes with the session on EVERY cleanup path (delete routes, the cron
|
||||
// and admin paths, scheduled-run teardown, error paths) — that is why it lives here
|
||||
// rather than in the two delete routes, where it left an entry behind, including up
|
||||
// to 4 KB of the user's buffered keystrokes.
|
||||
this.remoteWake?.drop(sessionId);
|
||||
|
||||
this.broadcast(SseEvent.SessionDeleted, { id: sessionId });
|
||||
}
|
||||
@@ -1596,6 +1657,23 @@ export class WebServer extends EventEmitter {
|
||||
'</head>',
|
||||
`<script>window.__codemanCliAvailable=${JSON.stringify(available)};</script>\n</head>`
|
||||
);
|
||||
// Which run modes the Run-menu picker (docs/custom-model-endpoints-plan.md) may
|
||||
// generate an entry for: read generically off the registry's `capabilities`
|
||||
// (never an id list here) so a CLI whose customModelInjection lands later shows
|
||||
// up in the picker with no frontend change, and one that ships `unsupported`
|
||||
// (antigravity, and `shell`'s `kind !== 'agent'`) never does.
|
||||
const customModelClis = enabledClis()
|
||||
.filter((entry) => entry.kind === 'agent' && entry.capabilities.customModelInjection.kind !== 'unsupported')
|
||||
.map((entry) => ({ id: entry.id, label: entry.label }));
|
||||
// Unlike the boolean-only __codemanCliAvailable above, this payload carries
|
||||
// `label`, a string a user's own clis.json can set (CliEntry.label, up to 60
|
||||
// chars) — see escapeScriptJson's own doc comment for why that needs escaping
|
||||
// and __codemanCliAvailable's booleans never did.
|
||||
const customModelClisJson = escapeScriptJson(JSON.stringify(customModelClis));
|
||||
html = html.replace(
|
||||
'</head>',
|
||||
`<script>window.__codemanCustomModelClis=${customModelClisJson};</script>\n</head>`
|
||||
);
|
||||
}
|
||||
if (!soloSessionId && process.env.CODEMAN_GESTURE === '1') {
|
||||
html = html.replace('</head>', `<script>window.__codemanGestureAvailable=true;</script>\n</head>`);
|
||||
@@ -1728,6 +1806,10 @@ export class WebServer extends EventEmitter {
|
||||
getStore: () => this.store,
|
||||
registerAttachment: (id: string, filePath: string, source: 'external' | 'codex-generated') =>
|
||||
this.registerAttachment(id, filePath, source),
|
||||
updateSessionName: (id: string, name: string) => this.mux.updateSessionName(id, name),
|
||||
// Opt-in: the first prompt lands in the tab name, mux-sessions.json, every
|
||||
// session:updated broadcast and /api/search, so it is a choice, not a default.
|
||||
isAutoNameEnabled: async () => (await this.readSettings()).autoNameSessions === true,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -2322,11 +2404,20 @@ export class WebServer extends EventEmitter {
|
||||
'scheduled:',
|
||||
'team:',
|
||||
'case:',
|
||||
'remote:',
|
||||
'custom-model:',
|
||||
];
|
||||
if (SESSION_PREFIXES.some((p) => event.startsWith(p))) {
|
||||
const d = (data ?? {}) as { sessionId?: string; id?: string; session?: { id?: string } };
|
||||
const d = (data ?? {}) as { sessionId?: string; id?: string; session?: { id?: string }; username?: string };
|
||||
const sessionId = d.sessionId ?? d.id ?? d.session?.id;
|
||||
const owner = sessionId ? this.sessions.get(sessionId)?.owner : undefined;
|
||||
// `remote:hostWaking` / `remote:hostWakeFailed` for a create/attach wake have no
|
||||
// session yet (nothing exists until the host is up), so the registry names the
|
||||
// requesting user instead; the payload carries `hostId`/`label`, which non-admins
|
||||
// are not shown elsewhere. No session and no requester: admins only (fail closed).
|
||||
if (!sessionId && event.startsWith('remote:') && d.username) {
|
||||
return { username: d.username, sessionScoped: true };
|
||||
}
|
||||
return { owner, sessionScoped: true };
|
||||
}
|
||||
// #20/#38: clipboard:write writes into the receiver's OS clipboard — route it to
|
||||
@@ -2699,6 +2790,64 @@ export class WebServer extends EventEmitter {
|
||||
});
|
||||
}
|
||||
|
||||
// Custom Model Endpoint Profiles (docs/custom-model-endpoints-plan.md): keeps
|
||||
// each saved endpoint's discovered model list current with no manual
|
||||
// "Discover" click, so a model added on the server side (or one that drops
|
||||
// off) shows up in the Run-menu picker within one cycle. Best-effort per
|
||||
// endpoint (refreshAllCustomModelHosts skips one that's unreachable rather
|
||||
// than failing the sweep) and off in tests for the same reason the Codex
|
||||
// poll above is — no real network to hit, no server instance to keep alive.
|
||||
if (!this.testMode) {
|
||||
this.cleanup.setInterval(
|
||||
() => {
|
||||
// Reads the setting fresh on every tick, same reasoning as
|
||||
// readPlanUsageTelemetryEnabled() beside it: a live toggle takes effect
|
||||
// on the very next cycle, not just at server boot, and turning the
|
||||
// feature off actually stops the polling instead of only hiding the UI.
|
||||
void readCustomModelEndpointsEnabled()
|
||||
.then((enabled) => {
|
||||
if (!enabled) return;
|
||||
return refreshAllCustomModelHosts();
|
||||
})
|
||||
.catch((err) => {
|
||||
console.error('[custom-model] periodic re-discovery failed:', getErrorMessage(err));
|
||||
});
|
||||
},
|
||||
CUSTOM_MODEL_REDISCOVER_INTERVAL_MS,
|
||||
{ description: 'custom model endpoint re-discovery' }
|
||||
);
|
||||
}
|
||||
|
||||
// Custom Model Endpoint Profiles: the swap-conflict check on the apply/create routes
|
||||
// only ever runs at THAT session's own launch/apply moment — it cannot catch a LATER
|
||||
// eviction triggered by a different session's normal use, since llama-swap has no push
|
||||
// notification of its own and only swaps in response to a real inference request
|
||||
// (confirmed live: a session created while nothing else conflicted at that instant can
|
||||
// still get silently displaced afterward). This periodic sweep is what catches that
|
||||
// case after the fact and tells the displaced session's user, rather than leaving them
|
||||
// to discover it only when their next prompt behaves unexpectedly.
|
||||
if (!this.testMode) {
|
||||
this.cleanup.setInterval(
|
||||
() => {
|
||||
detectCustomModelSwapDisplacements(this.sessions.values(), this._customModelDisplacedNotified)
|
||||
.then((displacements) => {
|
||||
for (const displacement of displacements) {
|
||||
this.broadcast(SseEvent.CustomModelSwappedOut, displacement);
|
||||
}
|
||||
})
|
||||
.catch((err) => {
|
||||
console.error('[custom-model] swap-displacement check failed:', getErrorMessage(err));
|
||||
});
|
||||
// Same cadence, unrelated concern: close any /api/events tail (see
|
||||
// getLatestLlamaSwapLogLine) nothing has polled in a while, so a loading banner
|
||||
// that finished (or was abandoned) doesn't leave a connection open forever.
|
||||
pruneIdleLlamaSwapLogTails();
|
||||
},
|
||||
CUSTOM_MODEL_SWAP_CHECK_INTERVAL_MS,
|
||||
{ description: 'custom model swap-displacement check' }
|
||||
);
|
||||
}
|
||||
|
||||
// Start scheduled runs cleanup timer
|
||||
this.cleanup.setInterval(
|
||||
() => {
|
||||
@@ -2853,6 +3002,229 @@ export class WebServer extends EventEmitter {
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Work out what a host reboot destroyed, and leave it on offer for the board.
|
||||
*
|
||||
* Runs inside `restoreMuxSessions()`, in the window after `reconcileSessions()`
|
||||
* has reported the dead sessions and before `finalizeRestoredState()` prunes
|
||||
* their records, so `state.json` is still the full picture here. That window is
|
||||
* the only place the plan can be built, which is why the boot pass builds it
|
||||
* even though nothing is rebuilt until a user clicks.
|
||||
*
|
||||
* Nothing is created here. The plan goes to `rebootRestoreRegistry`, the board
|
||||
* offers it as a banner, and `web/routes/reboot-restore-routes` rebuilds what
|
||||
* the user asks for. A wrong reboot guess therefore costs a line of text the
|
||||
* user dismisses, not N CLI processes nobody asked for.
|
||||
*
|
||||
* @returns how many sessions are on offer.
|
||||
*/
|
||||
private planRebootRestoreOffer(dead: string[], livePaneCount: number): number {
|
||||
if (dead.length === 0) return 0;
|
||||
|
||||
const persisted = this.store.getSessions();
|
||||
if (
|
||||
!looksLikeHostReboot({
|
||||
livePaneCount,
|
||||
deadSessionCount: dead.length,
|
||||
uptimeSeconds: osUptime(),
|
||||
newestPersistedActivityAt: newestPersistedActivity(persisted),
|
||||
now: Date.now(),
|
||||
})
|
||||
) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
const { restore, skipped } = planRebootRestore(dead, persisted, (workingDir) => existsSync(workingDir));
|
||||
if (skipped.length > 0) {
|
||||
console.log(`[Server] Reboot restore is passing over ${skipped.length} dead session(s):`);
|
||||
for (const rejection of skipped) {
|
||||
console.log(`[Server] ${rejection.sessionId}: ${rejection.reason}`);
|
||||
}
|
||||
}
|
||||
rebootRestoreRegistry.set(restore);
|
||||
if (restore.length > 0) {
|
||||
console.log(`[Server] Host reboot detected; offering ${restore.length} session(s) for restore`);
|
||||
}
|
||||
return restore.length;
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-apply the persisted state that a `Session` constructor does not take.
|
||||
*
|
||||
* The reboot-restore route builds a session from a record rather than
|
||||
* attaching to a surviving pane, so everything the constructor has no
|
||||
* parameter for starts at its default. Persisting such a session writes
|
||||
* `toState()` wholesale, which would REPLACE the record with the reduced
|
||||
* version — and for a pinned session that is worse than losing a setting,
|
||||
* because `cleanupSessionsByIds()` keeps a record only while it is pinned, so
|
||||
* dropping the pin hands the record to the next stale sweep.
|
||||
*
|
||||
* Split in two phases because the two halves have opposite timing needs:
|
||||
*
|
||||
* - `before-spawn` shapes the pane itself, so it has to land before the CLI
|
||||
* process starts, and before `setupSessionListeners()`, which reads the
|
||||
* image-watcher flag. The custom-model selection is an environment injection
|
||||
* and the nice priority is applied to the spawn.
|
||||
* - `after-spawn` is the session's own accumulated history. It must NOT land
|
||||
* on a session whose pane failed to start: the totals would then belong to a
|
||||
* session that never ran, and any later cleanup would add them to the
|
||||
* lifetime figures a second time.
|
||||
*
|
||||
* Respawn and Ralph are deliberately NOT re-armed: a machine that just came up
|
||||
* is the worst moment to turn an autonomous run loose, and the user re-arms
|
||||
* what they want. Ralph's loop CONFIGURATION does not survive either, because
|
||||
* `toState()` reads `ralphEnabled` and the completion phrase off a live
|
||||
* tracker, and there is no way to hold them without arming the loop.
|
||||
*/
|
||||
async reapplyPersistedSessionState(
|
||||
session: Session,
|
||||
saved: SessionState,
|
||||
phase: 'before-spawn' | 'after-spawn',
|
||||
options?: { rearmAutoResumeSchedule?: boolean }
|
||||
): Promise<void> {
|
||||
if (phase === 'before-spawn') {
|
||||
// The custom-model env has to be rebuilt from the endpoint store: the persist
|
||||
// deliberately keeps the injected VALUES out of state.json, so only the
|
||||
// bookkeeping survives a restart and the values are re-derived here.
|
||||
const savedCustomModel = (saved as { __customModel?: CustomModelBookkeeping }).__customModel;
|
||||
if (savedCustomModel) {
|
||||
session.setCustomModel(savedCustomModel, await this._rebuildCustomModelEnv(session, savedCustomModel));
|
||||
}
|
||||
if (saved.niceEnabled !== undefined || saved.niceValue !== undefined) {
|
||||
session.setNice({ enabled: saved.niceEnabled, niceValue: saved.niceValue });
|
||||
}
|
||||
// `setupSessionListeners()` READS this flag to decide whether to start the
|
||||
// watcher, so setting it later would leave the session reporting the feature
|
||||
// as on with nothing watching.
|
||||
if (saved.imageWatcherEnabled !== undefined) session.imageWatcherEnabled = saved.imageWatcherEnabled;
|
||||
return;
|
||||
}
|
||||
|
||||
if (saved.pinned) session.restorePin(true, saved.pinnedAt);
|
||||
if (saved.autoCompactEnabled !== undefined || saved.autoCompactThreshold !== undefined) {
|
||||
session.setAutoCompact(saved.autoCompactEnabled ?? false, saved.autoCompactThreshold, saved.autoCompactPrompt);
|
||||
}
|
||||
if (saved.autoClearEnabled !== undefined || saved.autoClearThreshold !== undefined) {
|
||||
session.setAutoClear(saved.autoClearEnabled ?? false, saved.autoClearThreshold);
|
||||
}
|
||||
if (saved.autoResumeEnabled) {
|
||||
// The stamp is re-armed by default, because a Codeman restart leaves the
|
||||
// limit footer un-reprinted and dropping it there would strand the pause.
|
||||
// A reboot restore opts out: that stamp predates the reboot, the pane is
|
||||
// new, and honouring it means every session the user restored types
|
||||
// `continue` into itself about a minute later, unattended. The setting
|
||||
// itself stays on either way, so it re-arms on the next limit message.
|
||||
const rearm = options?.rearmAutoResumeSchedule !== false;
|
||||
session.restoreAutoResume(true, rearm ? saved.autoResumeAt : undefined);
|
||||
}
|
||||
if (saved.inputTokens !== undefined || saved.outputTokens !== undefined || saved.totalCost !== undefined) {
|
||||
session.restoreTokens(saved.inputTokens ?? 0, saved.outputTokens ?? 0, saved.totalCost ?? 0);
|
||||
// Seed the daily-usage baseline, or the restored totals are counted again as new usage.
|
||||
this.lastRecordedTokens.set(session.id, {
|
||||
input: saved.inputTokens ?? 0,
|
||||
output: saved.outputTokens ?? 0,
|
||||
});
|
||||
}
|
||||
if (saved.color) session.setColor(saved.color);
|
||||
if (saved.flickerFilterEnabled !== undefined) session.flickerFilterEnabled = saved.flickerFilterEnabled;
|
||||
}
|
||||
|
||||
/**
|
||||
* Undo a session that was registered but never got a working pane.
|
||||
*
|
||||
* Deliberately NOT `cleanupSession()`, which is the user-initiated delete: that
|
||||
* path adds the session's token totals to the lifetime figures, demotes a
|
||||
* pinned record to `stopped` (the durable marker of an intentional kill, which
|
||||
* would make the session permanently ineligible for a reboot restore), drops
|
||||
* the persisted Ralph state, and recursively removes `.claude-images` from the
|
||||
* WORKING DIRECTORY, which belongs to the workspace rather than to this session
|
||||
* and may hold another live session's pasted images.
|
||||
*
|
||||
* Everything else `_doCleanupSession()` does, this has to do as well. It is the
|
||||
* inverse of `registerSessionWithLayout()` plus `setupSessionListeners()`, and
|
||||
* every registration those two make has to come back out — above all
|
||||
* `sessionListenerRefs`, whose presence makes `setupSessionListeners()` return
|
||||
* early. Leaving that entry behind is worse than the leak this function exists
|
||||
* to prevent: the retry reuses the same session id, wires no listeners at all,
|
||||
* and the user gets a tab that never shows output.
|
||||
*
|
||||
* The persisted record, the lifetime totals, the stored Ralph state and the
|
||||
* workspace's own files are left exactly as they were, so the session stays
|
||||
* restorable on the next attempt.
|
||||
*/
|
||||
async discardPartiallyBuiltSession(sessionId: string): Promise<void> {
|
||||
const session = this.sessions.get(sessionId);
|
||||
if (!session) return;
|
||||
this.sessions.delete(sessionId);
|
||||
|
||||
// --- the inverse of setupSessionListeners(), in reverse order ---
|
||||
// Listeners first: while they are attached, one of them can still reach a
|
||||
// tracker this is about to stop.
|
||||
const listeners = this.sessionListenerRefs.get(sessionId);
|
||||
if (listeners) {
|
||||
detachSessionListeners(session, listeners);
|
||||
this.sessionListenerRefs.delete(sessionId);
|
||||
}
|
||||
// An FSWatcher on the workspace that nothing else closes.
|
||||
imageWatcher.unwatchSession(sessionId);
|
||||
// An fs.watch on the workspace (or on @fix_plan.md), likewise.
|
||||
session.ralphTracker.stopWatchingFixPlan();
|
||||
const summaryTracker = this.runSummaryTrackers.get(sessionId);
|
||||
if (summaryTracker) {
|
||||
// Closes the run's own record before the tracker goes, the way
|
||||
// `_doCleanupSession()` does. Cosmetic rather than load-bearing, but a
|
||||
// run left open reads as still going in the away digest.
|
||||
summaryTracker.recordSessionStopped();
|
||||
summaryTracker.stop();
|
||||
this.runSummaryTrackers.delete(sessionId);
|
||||
}
|
||||
// Also mirrors `_doCleanupSession()`. The PERSISTED Ralph state is left
|
||||
// alone on purpose (that is one of the things separating this from
|
||||
// cleanupSession); this only clears the in-memory tracker the failed
|
||||
// construction built, which the retry reuses the id of.
|
||||
session.ralphTracker.fullReset();
|
||||
|
||||
// --- what anything else may have attached to this id in the meantime ---
|
||||
// A rebuild can fail AFTER startInteractive() resolved, and a restored
|
||||
// workspace still carries Codeman's hooks, so the CLI can post a hook event
|
||||
// within milliseconds. Each of these outlives the listeners and would
|
||||
// otherwise meet the retry, which reuses the same session id by design.
|
||||
this.stopTranscriptWatcher(sessionId);
|
||||
attachmentRegistry.clearSession(sessionId);
|
||||
sessionWaits.notifySignal(sessionId, 'exit');
|
||||
sessionWaits.cancelAll(sessionId);
|
||||
approvalInbox.resolveForSession(sessionId, 'session_ended');
|
||||
|
||||
// --- the inverse of the construction itself ---
|
||||
this.sse.cleanupSessionBatches(sessionId);
|
||||
this.persistDeb.cancelKey(sessionId);
|
||||
fileStreamManager.closeSessionStreams(sessionId);
|
||||
// `lastRecordedTokens` is deliberately NOT deleted: the `after-spawn` phase
|
||||
// seeds it as the daily-usage baseline for these restored totals, and the
|
||||
// retry reuses the id, so dropping it would count them as new usage.
|
||||
// The per-session custom-model config dir carries the endpoint's API key, and
|
||||
// `before-spawn` may already have written it. Nothing else would ever remove
|
||||
// it: the stale sweep only touches state.json. A retry rewrites it.
|
||||
removeConfigDir(customModelConfigDir(sessionId));
|
||||
try {
|
||||
session.removeAllListeners();
|
||||
await session.stop(true);
|
||||
} catch (err) {
|
||||
console.warn(`[Server] stopping a partially built session failed: ${getErrorMessage(err)}`);
|
||||
// `stop()` kills the mux session in its last block, after destroying its
|
||||
// trackers, so a throw on the way there leaves the pane running.
|
||||
await this.mux.killSession(sessionId).catch(() => {});
|
||||
}
|
||||
try {
|
||||
await this.tabLayouts.sessionsRemoved([{ id: sessionId, owner: session.owner }]);
|
||||
} catch (err) {
|
||||
console.warn(`[Server] releasing the tab layout slot failed: ${getErrorMessage(err)}`);
|
||||
}
|
||||
// Any `session:updated` the half-built session emitted before it failed left a
|
||||
// tab on every other open board, and the client's handler is an upsert.
|
||||
this.broadcast(SseEvent.SessionDeleted, { id: sessionId });
|
||||
}
|
||||
|
||||
private async restoreMuxSessions(): Promise<boolean> {
|
||||
try {
|
||||
// Reconcile mux sessions to find which ones are still alive (also discovers unknown ones)
|
||||
@@ -2862,11 +3234,30 @@ export class WebServer extends EventEmitter {
|
||||
console.log(`[Server] Discovered ${discovered.length} unknown mux session(s)`);
|
||||
}
|
||||
|
||||
// Build the reboot-restore offer HERE: `dead` is only known after
|
||||
// reconciliation, and the records it reads are pruned by
|
||||
// `cleanupStaleSessions()` as soon as `finalizeRestoredState()` runs.
|
||||
//
|
||||
// Guarded on its own, because this runs inside the try that decides whether
|
||||
// RECOVERY succeeded. A throw here would otherwise be caught below, report
|
||||
// restoration as failed, and block the stale cleanup and layout
|
||||
// reconciliation that follow — turning an optional convenience into a
|
||||
// failure of the thing it is supposed to help. An offer nobody gets is the
|
||||
// correct way for this to fail.
|
||||
try {
|
||||
this.planRebootRestoreOffer(dead, alive.length);
|
||||
} catch (err) {
|
||||
console.error('[Server] Building the reboot-restore offer failed; continuing recovery:', err);
|
||||
}
|
||||
|
||||
if (alive.length > 0 || discovered.length > 0) {
|
||||
console.log(`[Server] Found ${alive.length + discovered.length} alive mux session(s) from previous run`);
|
||||
|
||||
// For each alive mux session, create a Session object if it doesn't exist
|
||||
const muxSessions = this.mux.getSessions();
|
||||
// Host-level config lives in remote-hosts.json, not in the persisted session
|
||||
// snapshot, so refresh the fields that only exist there (see the helper).
|
||||
const remoteHostsById = new Map((await readRemoteHosts(getDataDir())).map((host) => [host.id, host]));
|
||||
for (const muxSession of muxSessions) {
|
||||
if (!this.sessions.has(muxSession.sessionId)) {
|
||||
// Restore session settings from state.json (single source of truth)
|
||||
@@ -2903,6 +3294,7 @@ export class WebServer extends EventEmitter {
|
||||
workingDir: muxSession.workingDir,
|
||||
mode: muxSession.mode,
|
||||
name: sessionName,
|
||||
nameSource: savedState?.nameSource,
|
||||
// When the session FIRST started, not when this server booted.
|
||||
// Without it every recovered session was restamped `Date.now()` on
|
||||
// each restart, so a week-old pane read as "created 2m ago" on the
|
||||
@@ -2943,7 +3335,9 @@ export class WebServer extends EventEmitter {
|
||||
// respawn rebuilds a LOCAL command, breaking the pane and silently
|
||||
// erasing `remote` from state.json on the next persist. mux-sessions.json
|
||||
// round-trips MuxSession.remote; state.json carries SessionState.remote.
|
||||
remote: muxSession.remote ?? savedState?.remote,
|
||||
// Host-level fields are refreshed from remote-hosts.json on top, or a
|
||||
// field added to the host config after launch would never arrive.
|
||||
remote: rehydrateRemoteHostFields(muxSession.remote ?? savedState?.remote, remoteHostsById),
|
||||
// Docker metadata round-trips the same way (mux-sessions.json carries
|
||||
// MuxSession.docker; state.json carries SessionState.docker), so recovery
|
||||
// rebuilds the `docker exec` launch instead of a broken local command.
|
||||
@@ -3326,6 +3720,10 @@ export class WebServer extends EventEmitter {
|
||||
// got wrong once.
|
||||
void stopDeepSeekWeb();
|
||||
|
||||
// Same teardown rule: the per-endpoint llama-swap log tails are otherwise closed
|
||||
// only by the periodic idle sweep, whose interval is disposed just below.
|
||||
closeAllLlamaSwapLogTails();
|
||||
|
||||
// Dispose all managed timers (intervals + resettable timeouts)
|
||||
this.cleanup.dispose();
|
||||
|
||||
@@ -3337,6 +3735,10 @@ export class WebServer extends EventEmitter {
|
||||
// response), so without this a 10-minute wait holds shutdown open.
|
||||
sessionWaits.cancelEverything();
|
||||
approvalInbox.stop();
|
||||
// Same reason as `cancelEverything` above: an in-flight wake is awaited by a request,
|
||||
// and `app.close()` (the last line of this method) does not abort in-flight requests —
|
||||
// so without this a restart during a wake waits out the readiness poll.
|
||||
this.remoteWake?.stop();
|
||||
|
||||
this.lastRecordedTokens.clear();
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
*
|
||||
* Extracted from server.ts for modularity. Provides:
|
||||
* - `SessionListenerRefs` interface (named listener references for leak-free cleanup)
|
||||
* - `createSessionListeners()` — builds all 25 listener handlers via dependency injection
|
||||
* - `createSessionListeners()` — builds all session listener handlers via dependency injection
|
||||
* - `attachSessionListeners()` / `detachSessionListeners()` — symmetric attach/detach
|
||||
*
|
||||
* The detach function deduplicates a pattern that was previously copy-pasted 3 times
|
||||
@@ -29,6 +29,8 @@ import { getLifecycleLog } from '../session-lifecycle-log.js';
|
||||
import { fileStreamManager } from '../file-stream-manager.js';
|
||||
import { sessionWaits } from './session-wait-registry.js';
|
||||
import { approvalInbox } from './approval-inbox.js';
|
||||
import { composeAutoSessionName, deriveAutoSessionName } from '../session-auto-name.js';
|
||||
import { MAX_SESSION_NAME_LENGTH } from '../config/terminal-limits.js';
|
||||
|
||||
/** Stored listener references for session cleanup (prevents memory leaks) */
|
||||
export interface SessionListenerRefs {
|
||||
@@ -63,6 +65,7 @@ export interface SessionListenerRefs {
|
||||
bashToolEnd: (tool: ActiveBashTool) => void;
|
||||
bashToolsUpdate: (tools: ActiveBashTool[]) => void;
|
||||
attachmentRequested: (event: { path: string; source: 'external' | 'codex-generated' }) => void;
|
||||
promptSubmitted: (prompt: string) => void;
|
||||
}
|
||||
|
||||
/** Dependencies injected by WebServer — keeps listener creation decoupled from server internals. */
|
||||
@@ -83,10 +86,13 @@ interface SessionListenerDeps {
|
||||
cleanupRespawnOnExit(sessionId: string): void;
|
||||
getStore(): import('../state-store.js').StateStore;
|
||||
registerAttachment(sessionId: string, filePath: string, source: 'external' | 'codex-generated'): Promise<void>;
|
||||
updateSessionName(sessionId: string, name: string): boolean;
|
||||
/** The synced `autoNameSessions` setting, read fresh so a flip applies to the next prompt. */
|
||||
isAutoNameEnabled(): Promise<boolean>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates all 26 session listener handlers, capturing dependencies via closure.
|
||||
* Creates all session listener handlers, capturing dependencies via closure.
|
||||
* Call `attachSessionListeners()` after to wire them to the session.
|
||||
*/
|
||||
export function createSessionListeners(session: Session, deps: SessionListenerDeps): SessionListenerRefs {
|
||||
@@ -451,6 +457,30 @@ export function createSessionListeners(session: Session, deps: SessionListenerDe
|
||||
console.error(`[Attachment] Failed to register ${event.path} for ${session.id}:`, err);
|
||||
});
|
||||
},
|
||||
|
||||
/**
|
||||
* Names a placeholder tab after its first real prompt (`w3-case: fix the
|
||||
* login redirect`), behind the synced `autoNameSessions` setting. The
|
||||
* eligibility check comes first so the settings read costs nothing on the
|
||||
* prompts of an already-named session; a prompt that yields no title (a
|
||||
* slash command) leaves the session eligible for the next one.
|
||||
*/
|
||||
promptSubmitted: (prompt: string) => {
|
||||
if (session.nameSource !== 'placeholder') return;
|
||||
const title = deriveAutoSessionName(prompt);
|
||||
if (!title) return;
|
||||
void deps
|
||||
.isAutoNameEnabled()
|
||||
.then((enabled) => {
|
||||
if (!enabled) return;
|
||||
const name = composeAutoSessionName(session.name, title, MAX_SESSION_NAME_LENGTH);
|
||||
if (!session.applyAutoName(name)) return;
|
||||
deps.updateSessionName(session.id, session.name);
|
||||
deps.persistSessionState(session);
|
||||
deps.broadcast(SseEvent.SessionUpdated, deps.getSessionStateWithRespawn(session));
|
||||
})
|
||||
.catch((err) => console.error(`[Session] auto-name failed for ${session.id}:`, err));
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
@@ -487,6 +517,7 @@ export function attachSessionListeners(session: Session, refs: SessionListenerRe
|
||||
session.on('bashToolEnd', refs.bashToolEnd);
|
||||
session.on('bashToolsUpdate', refs.bashToolsUpdate);
|
||||
session.on('attachmentRequested', refs.attachmentRequested);
|
||||
session.on('promptSubmitted', refs.promptSubmitted);
|
||||
}
|
||||
|
||||
/** Detach all listeners from a session (prevents memory leaks from closure references). */
|
||||
@@ -522,4 +553,5 @@ export function detachSessionListeners(session: Session, refs: SessionListenerRe
|
||||
session.off('bashToolEnd', refs.bashToolEnd);
|
||||
session.off('bashToolsUpdate', refs.bashToolsUpdate);
|
||||
session.off('attachmentRequested', refs.attachmentRequested);
|
||||
session.off('promptSubmitted', refs.promptSubmitted);
|
||||
}
|
||||
|
||||
+33
-3
@@ -5,7 +5,7 @@
|
||||
* and referenced by the frontend (`SSE_EVENTS` in `constants.js`).
|
||||
* Both files MUST be kept in sync.
|
||||
*
|
||||
* 158 event constants organized by category:
|
||||
* 161 event constants organized by category:
|
||||
* - **Core** (1): init
|
||||
* - **Transport** (1): sse:heartbeat
|
||||
* - **Session lifecycle** (23): created, updated, deleted, terminal, idle, working, ...
|
||||
@@ -14,7 +14,7 @@
|
||||
* - **Session: Plan** (4): planTaskUpdate, planCheckpoint, planRollback, planTaskAdded
|
||||
* - **Tasks** (4): created, completed, failed, updated
|
||||
* - **Mux** (4): created, killed, died, statsUpdated
|
||||
* - **Remote auto-reconnect** (3): sessionDropped, sessionReconnected, reconnectExhausted
|
||||
* - **Remote auto-reconnect / wake** (5): sessionDropped, sessionReconnected, reconnectExhausted, hostWaking, hostWakeFailed
|
||||
* - **Respawn** (24): stateChanged, cycleStarted/Completed, step*, aiCheck*, planCheck*, timer*, log, ...
|
||||
* - **Subagents** (7): discovered, updated, tool_call, tool_result, progress, message, completed
|
||||
* - **Workflow runs** (3): run_discovered, run_updated, run_removed (ultracode / Workflow tool)
|
||||
@@ -28,6 +28,7 @@
|
||||
* - **Hooks** (10): idle_prompt, permission_prompt, elicitation_dialog, elicitation_complete, elicitation_response, stop, agent_working, teammate_idle, task_completed, prompt_submitted
|
||||
* (agent_working is the odd one out: reported by the DeepSeek Harness status bridge, not by a Claude Code hook)
|
||||
* - **Approvals** (3): pending, updated, resolved (cross-session Approvals Inbox)
|
||||
* - **Custom Model Endpoint Profiles** (1): swapped-out (a session's model got evicted by another session on the same llama-swap endpoint)
|
||||
* - **Orchestrator** (12): stateChanged, planProgress, planReady, phase*, verification, task*, completed, error
|
||||
* - **Clipboard** (1): write
|
||||
* - **Cases** (4): created, linked, deleted, order-changed
|
||||
@@ -176,7 +177,9 @@ export const MuxDied = 'mux:died' as const;
|
||||
/** tmux session stats refreshed. */
|
||||
export const MuxStatsUpdated = 'mux:statsUpdated' as const;
|
||||
|
||||
// ─── Remote auto-reconnect (COD-108) ─────────────────────────────────────────
|
||||
// ─── Remote auto-reconnect (COD-108) + wake-on-LAN ───────────────────────────
|
||||
// Session-scoped in multi-user mode (`deriveSseHint`): routed to the session's owner,
|
||||
// or — for a wake with no session yet — to the requesting `username` in the payload.
|
||||
|
||||
/** A remote session's local ssh pane died; an auto-reconnect attempt is starting. */
|
||||
export const RemoteSessionDropped = 'remote:sessionDropped' as const;
|
||||
@@ -184,6 +187,15 @@ export const RemoteSessionDropped = 'remote:sessionDropped' as const;
|
||||
export const RemoteSessionReconnected = 'remote:sessionReconnected' as const;
|
||||
/** Auto-reconnect gave up after the bounded backoff cap — manual reconnect needed. */
|
||||
export const RemoteReconnectExhausted = 'remote:reconnectExhausted' as const;
|
||||
/**
|
||||
* User input arrived for a session whose host is unreachable, so a Wake-on-LAN
|
||||
* command was started (see `remote-wake.ts`). Input sent meanwhile is buffered.
|
||||
* Payload: `sessionId` (session wake) or `forNewSession: true` + `username`
|
||||
* (create/attach wake), `hostId`, `label`, `queuedInput`.
|
||||
*/
|
||||
export const RemoteHostWaking = 'remote:hostWaking' as const;
|
||||
/** The host did not come back within the wake timeout — buffered input is still held. */
|
||||
export const RemoteHostWakeFailed = 'remote:hostWakeFailed' as const;
|
||||
|
||||
// ─── Respawn ─────────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -384,6 +396,19 @@ export const ApprovalUpdated = 'approval:updated' as const;
|
||||
/** A pending approval left the inbox (answered, superseded, expired, ...). */
|
||||
export const ApprovalResolved = 'approval:resolved' as const;
|
||||
|
||||
// ─── Custom Model Endpoint Profiles ──────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* A session's own custom-model selection is no longer the model llama-swap has loaded —
|
||||
* ANOTHER session's activity on the same endpoint evicted it (llama.cpp/llama-swap runs
|
||||
* one model at a time). Detected after the fact by a periodic sweep (`server.ts`), never
|
||||
* at the moment of eviction itself, since llama-swap has no push notification of its own;
|
||||
* this session's next prompt will trigger reloading its model, evicting whatever displaced
|
||||
* it in turn. Fires at most once per displacement (cleared once the sweep sees the
|
||||
* session's own model loaded again), so it can't spam on every sweep interval.
|
||||
*/
|
||||
export const CustomModelSwappedOut = 'custom-model:swapped-out' as const;
|
||||
|
||||
// ─── Orchestrator ────────────────────────────────────────────────────────────
|
||||
|
||||
/** Orchestrator state machine transitioned. */
|
||||
@@ -535,6 +560,8 @@ export const SseEvent = {
|
||||
RemoteSessionDropped,
|
||||
RemoteSessionReconnected,
|
||||
RemoteReconnectExhausted,
|
||||
RemoteHostWaking,
|
||||
RemoteHostWakeFailed,
|
||||
|
||||
// Respawn
|
||||
RespawnStarted,
|
||||
@@ -638,6 +665,9 @@ export const SseEvent = {
|
||||
ApprovalUpdated,
|
||||
ApprovalResolved,
|
||||
|
||||
// Custom Model Endpoint Profiles
|
||||
CustomModelSwappedOut,
|
||||
|
||||
// Orchestrator
|
||||
OrchestratorStateChanged,
|
||||
OrchestratorPlanProgress,
|
||||
|
||||
@@ -0,0 +1,533 @@
|
||||
/**
|
||||
* @fileoverview A capture drawn for a bigger pane makes the client replay once.
|
||||
*
|
||||
* A visible-frame capture repaints each row at an absolute position, counting
|
||||
* up to the PANE's height and out to the PANE's width. A terminal shorter than
|
||||
* that clamps every address past its own height onto its last line, so the
|
||||
* overflow rows overwrite one another and the rows underneath are lost. A
|
||||
* narrower terminal wraps every painted row, and the wrap on the last one
|
||||
* scrolls the whole frame up by one. The client cannot see either from the
|
||||
* escape sequence, so the terminal response reports the geometry the capture
|
||||
* was taken at (`captureCols`/`captureRows`) and `selectSession` replays once
|
||||
* at the size that stuck.
|
||||
*
|
||||
* The comparison runs on a `mux-visible` response ONLY. The other two sources
|
||||
* position no rows absolutely, so a size mismatch damages neither and a replay
|
||||
* repairs neither, and the last case here pins that the expensive one is left
|
||||
* alone.
|
||||
*
|
||||
* These drive the REAL client in chromium and stub only the terminal endpoint,
|
||||
* because the mismatch itself needs two viewports to stage against live tmux.
|
||||
* Without the fix the first assertion below sees one fetch instead of two.
|
||||
*
|
||||
* Port: 3252 (capture geometry retry)
|
||||
*
|
||||
* Run: npx vitest run --config config/vitest.browser.config.ts test/capture-geometry-retry.browser.test.ts
|
||||
*/
|
||||
|
||||
import { describe, it, expect, beforeAll, afterAll } from 'vitest';
|
||||
import { chromium, type Browser, type BrowserContext, type Page } from 'playwright';
|
||||
import { WebServer } from '../src/web/server.js';
|
||||
|
||||
const PORT = 3252;
|
||||
const BASE_URL = `http://localhost:${PORT}`;
|
||||
|
||||
let server: WebServer;
|
||||
let browser: Browser;
|
||||
|
||||
beforeAll(async () => {
|
||||
server = new WebServer(PORT, false, true); // testMode
|
||||
await server.start();
|
||||
browser = await chromium.launch({ headless: true });
|
||||
}, 60_000);
|
||||
|
||||
afterAll(async () => {
|
||||
await browser?.close();
|
||||
await server?.stop();
|
||||
}, 30_000);
|
||||
|
||||
/** A visible-frame capture: one absolutely-addressed paint per row. */
|
||||
function paneSnapshot(rows: number): string {
|
||||
const parts: string[] = [];
|
||||
for (let row = 1; row <= rows; row++) parts.push(`\x1b[${row};1Hprobe-row-${row}`);
|
||||
parts.push(`\x1b[${rows};6H`);
|
||||
return parts.join('');
|
||||
}
|
||||
|
||||
/**
|
||||
* Serve every terminal fetch from a stub reporting `captureRows`, counting the
|
||||
* fetches. The real route needs live tmux to produce a mismatched frame.
|
||||
*
|
||||
* `source` is DERIVED from the request the way the real route derives it: a
|
||||
* `full=1` request whose capture came back is `mux-full-history`, and every
|
||||
* other one is `mux-visible`. The route cannot answer `full=1` with
|
||||
* `mux-visible`, so a stub that did would stage a combination production never
|
||||
* produces, and a test resting on it would prove nothing about production. A
|
||||
* test that needs some other source passes it explicitly and says why.
|
||||
*/
|
||||
async function stubTerminal(
|
||||
page: Page,
|
||||
captureRows: number,
|
||||
counter: { n: number; urls: string[] },
|
||||
options: { source?: string; captureCols?: number } = {}
|
||||
) {
|
||||
const captureCols = options.captureCols ?? 200;
|
||||
await page.route('**/api/sessions/*/terminal*', async (route) => {
|
||||
const url = route.request().url();
|
||||
counter.n += 1;
|
||||
counter.urls.push(url);
|
||||
const source = options.source ?? (url.includes('full=1') ? 'mux-full-history' : 'mux-visible');
|
||||
await route.fulfill({
|
||||
status: 200,
|
||||
contentType: 'application/json',
|
||||
body: JSON.stringify({
|
||||
success: true,
|
||||
data: {
|
||||
terminalBuffer: paneSnapshot(captureRows),
|
||||
status: 'idle',
|
||||
fullSize: 1024,
|
||||
retainedBytes: 1024,
|
||||
truncated: false,
|
||||
truncationReason: null,
|
||||
source,
|
||||
captureCols,
|
||||
captureRows,
|
||||
},
|
||||
}),
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* As `stubTerminal`, but reading its geometry from a holder the test can change
|
||||
* between selects. That is what lets one case watch a pane stop fitting and
|
||||
* start fitting again, which a stub fixed at construction cannot show.
|
||||
*/
|
||||
async function stubTerminalDynamic(
|
||||
page: Page,
|
||||
counter: { n: number; urls: string[] },
|
||||
state: { captureRows: number; captureCols: number }
|
||||
) {
|
||||
await page.route('**/api/sessions/*/terminal*', async (route) => {
|
||||
const url = route.request().url();
|
||||
counter.n += 1;
|
||||
counter.urls.push(url);
|
||||
await route.fulfill({
|
||||
status: 200,
|
||||
contentType: 'application/json',
|
||||
body: JSON.stringify({
|
||||
success: true,
|
||||
data: {
|
||||
terminalBuffer: paneSnapshot(state.captureRows),
|
||||
status: 'idle',
|
||||
fullSize: 1024,
|
||||
retainedBytes: 1024,
|
||||
truncated: false,
|
||||
truncationReason: null,
|
||||
source: url.includes('full=1') ? 'mux-full-history' : 'mux-visible',
|
||||
captureCols: state.captureCols,
|
||||
captureRows: state.captureRows,
|
||||
},
|
||||
}),
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Answer every fetch with the geometry the client itself is asking for, read
|
||||
* live from the page. That is the clamp signature: `getTerminalDimensions()`
|
||||
* floors at 40x10 while `fitAddon.fit()` does not, so a small enough viewport
|
||||
* makes the pane permanently bigger than the terminal at a size the client
|
||||
* requested itself.
|
||||
*/
|
||||
async function stubTerminalAtRequestedSize(page: Page, counter: { n: number; urls: string[] }) {
|
||||
await page.route('**/api/sessions/*/terminal*', async (route) => {
|
||||
counter.n += 1;
|
||||
counter.urls.push(route.request().url());
|
||||
const dims = await page.evaluate(
|
||||
() =>
|
||||
(
|
||||
window as unknown as { app: { getTerminalDimensions?: () => { cols: number; rows: number } | null } }
|
||||
).app.getTerminalDimensions?.() ?? null
|
||||
);
|
||||
await route.fulfill({
|
||||
status: 200,
|
||||
contentType: 'application/json',
|
||||
body: JSON.stringify({
|
||||
success: true,
|
||||
data: {
|
||||
terminalBuffer: paneSnapshot(dims?.rows ?? 10),
|
||||
status: 'idle',
|
||||
fullSize: 1024,
|
||||
retainedBytes: 1024,
|
||||
truncated: false,
|
||||
truncationReason: null,
|
||||
source: 'mux-visible',
|
||||
captureCols: dims?.cols,
|
||||
captureRows: dims?.rows,
|
||||
},
|
||||
}),
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
/** The widest terminal this suite's 1280px viewport can produce, with margin. */
|
||||
const WIDER_THAN_ANY_TERMINAL_COLS = 500;
|
||||
|
||||
async function openSession(page: Page): Promise<string> {
|
||||
await page.goto(BASE_URL, { waitUntil: 'domcontentloaded' });
|
||||
await page.waitForFunction(() => document.body.classList.contains('app-loaded'), { timeout: 10_000 });
|
||||
// xterm is loaded from /vendor, so the terminal appears a beat after the app.
|
||||
// Without it `app.terminal.rows` reads 0 and every height comparison below
|
||||
// would pass vacuously.
|
||||
await page.waitForFunction(() => (window as unknown as { app?: { terminal?: unknown } }).app?.terminal, null, {
|
||||
timeout: 30_000,
|
||||
});
|
||||
return page.evaluate(async () => {
|
||||
const res = await fetch('/api/sessions', {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ workingDir: '/tmp', name: 'capture-geometry-test' }),
|
||||
});
|
||||
const body = await res.json();
|
||||
return body.data?.session?.id ?? body.data?.id ?? body.id;
|
||||
});
|
||||
}
|
||||
|
||||
/** The terminal is sized by the first select, so this only reads after one. */
|
||||
async function terminalRows(page: Page): Promise<number> {
|
||||
return page.evaluate(() => (window as unknown as { app: { terminal?: { rows: number } } }).app.terminal?.rows ?? 0);
|
||||
}
|
||||
|
||||
/** As above, for the width half of the comparison. */
|
||||
async function terminalCols(page: Page): Promise<number> {
|
||||
return page.evaluate(() => (window as unknown as { app: { terminal?: { cols: number } } }).app.terminal?.cols ?? 0);
|
||||
}
|
||||
|
||||
async function select(page: Page, sessionId: string, options: object = {}): Promise<void> {
|
||||
await page.evaluate(
|
||||
async ({ sid, opts }) => {
|
||||
const app = (window as unknown as { app: { selectSession: (id: string, o?: object) => Promise<void> } }).app;
|
||||
await app.selectSession(sid, opts);
|
||||
},
|
||||
{ sid: sessionId, opts: options }
|
||||
);
|
||||
await page.waitForTimeout(1500);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spend the per-page full-history allowance and forget what it cost. Every
|
||||
* geometry comparison below runs on a `mux-visible` response, and the route
|
||||
* only produces one for a request sent WITHOUT `full=1`, so reaching that shape
|
||||
* means not being the first select of the page — which is what a tab switch is.
|
||||
*/
|
||||
async function consumeFullHistory(
|
||||
page: Page,
|
||||
sessionId: string,
|
||||
counter: { n: number; urls: string[] }
|
||||
): Promise<void> {
|
||||
await select(page, sessionId);
|
||||
counter.n = 0;
|
||||
counter.urls.length = 0;
|
||||
}
|
||||
|
||||
async function closeSession(page: Page, sessionId: string): Promise<void> {
|
||||
await page.evaluate(
|
||||
(sid: string) => fetch(`/api/sessions/${sid}`, { method: 'DELETE' }).then(() => undefined),
|
||||
sessionId
|
||||
);
|
||||
}
|
||||
|
||||
describe('a capture bigger than the terminal', () => {
|
||||
let context: BrowserContext;
|
||||
let page: Page;
|
||||
|
||||
afterAll(async () => {
|
||||
await context?.close();
|
||||
});
|
||||
|
||||
it('replays once when the captured pane is taller, and stops at one retry', async () => {
|
||||
context = await browser.newContext({ viewport: { width: 1280, height: 800 } });
|
||||
page = await context.newPage();
|
||||
const sessionId = await openSession(page);
|
||||
expect(sessionId).toBeTruthy();
|
||||
|
||||
// 200 rows is taller than any terminal this viewport can produce, so the
|
||||
// trigger is the captured height alone and not a size that moved.
|
||||
const fetches = { n: 0, urls: [] as string[] };
|
||||
await stubTerminal(page, 200, fetches);
|
||||
// A tab switch is where a visible-frame response arrives, so that is what
|
||||
// this measures. The first select of the page takes the full-history path
|
||||
// and is covered by its own case below.
|
||||
await consumeFullHistory(page, sessionId, fetches);
|
||||
await select(page, sessionId, { forceReload: true });
|
||||
|
||||
// The terminal is sized by that select, so the premise is checkable now.
|
||||
expect(await terminalRows(page)).toBeLessThan(200);
|
||||
// One original load plus exactly one retry. `resizeRetry` caps it there:
|
||||
// the retry's own response reports the same mismatch, so an uncapped
|
||||
// implementation would loop.
|
||||
expect(fetches.n).toBe(2);
|
||||
|
||||
await closeSession(page, sessionId);
|
||||
await context.close();
|
||||
}, 60_000);
|
||||
|
||||
it('retries at the same scope the first pass used, not a wider one', async () => {
|
||||
// The retry re-arms the full-history flag only when the pass that ran had
|
||||
// consumed it. A tab switch takes the bounded tail, so its retry must take
|
||||
// the tail too; clearing the flag unconditionally would upgrade it into a
|
||||
// fresh multi-megabyte scrollback capture the user never asked for.
|
||||
context = await browser.newContext({ viewport: { width: 1280, height: 800 } });
|
||||
page = await context.newPage();
|
||||
const sessionId = await openSession(page);
|
||||
|
||||
const fetches = { n: 0, urls: [] as string[] };
|
||||
await stubTerminal(page, 200, fetches);
|
||||
|
||||
// First select: a fresh session, so this one pulls full history. It does
|
||||
// NOT retry, because the geometry comparison runs on a visible-frame
|
||||
// response and a `full=1` request cannot produce one.
|
||||
await select(page, sessionId);
|
||||
expect(fetches.n).toBe(1);
|
||||
expect(fetches.urls.filter((u) => u.includes('full=1'))).toHaveLength(1);
|
||||
|
||||
// Re-select the SAME session. `selectSession` early-returns on an already
|
||||
// active session unless forceReload is set, and forceReload is the shape a
|
||||
// tab switch back to this session takes: `_fullHistoryLoaded` still holds
|
||||
// it, so neither this pass nor its retry asks for full history again.
|
||||
await select(page, sessionId, { forceReload: true });
|
||||
const tabSwitchUrls = fetches.urls.slice(1);
|
||||
expect(tabSwitchUrls.length).toBe(2);
|
||||
expect(tabSwitchUrls.filter((u) => u.includes('full=1'))).toHaveLength(0);
|
||||
|
||||
await closeSession(page, sessionId);
|
||||
await context.close();
|
||||
}, 60_000);
|
||||
|
||||
it('does not replay when the captured pane fits the terminal', async () => {
|
||||
context = await browser.newContext({ viewport: { width: 1280, height: 800 } });
|
||||
page = await context.newPage();
|
||||
const sessionId = await openSession(page);
|
||||
|
||||
// Five rows is shorter than any terminal this viewport can produce, so the
|
||||
// frame fits, nothing is clamped, and nothing needs repeating. A retry here
|
||||
// would double the work of every tab switch.
|
||||
const fetches = { n: 0, urls: [] as string[] };
|
||||
await stubTerminal(page, 5, fetches, { captureCols: 40 });
|
||||
await consumeFullHistory(page, sessionId, fetches);
|
||||
await select(page, sessionId, { forceReload: true });
|
||||
|
||||
expect(await terminalRows(page)).toBeGreaterThan(5);
|
||||
expect(fetches.n).toBe(1);
|
||||
|
||||
await closeSession(page, sessionId);
|
||||
await context.close();
|
||||
}, 60_000);
|
||||
|
||||
it('replays once when the captured pane is wider', async () => {
|
||||
// A pane wider than the terminal damages the same frame a second way.
|
||||
// `formatPaneSnapshot` paints every row out to the PANE's width, so a
|
||||
// narrower browser wraps each painted row, and the wrap on the last row
|
||||
// scrolls the whole frame up by one. The height here fits deliberately, so
|
||||
// the width is the only thing that can trigger the replay.
|
||||
context = await browser.newContext({ viewport: { width: 1280, height: 800 } });
|
||||
page = await context.newPage();
|
||||
const sessionId = await openSession(page);
|
||||
|
||||
const fetches = { n: 0, urls: [] as string[] };
|
||||
await stubTerminal(page, 5, fetches, { captureCols: WIDER_THAN_ANY_TERMINAL_COLS });
|
||||
await consumeFullHistory(page, sessionId, fetches);
|
||||
await select(page, sessionId, { forceReload: true });
|
||||
|
||||
expect(await terminalRows(page)).toBeGreaterThan(5);
|
||||
expect(await terminalCols(page)).toBeLessThan(WIDER_THAN_ANY_TERMINAL_COLS);
|
||||
expect(fetches.n).toBe(2);
|
||||
|
||||
await closeSession(page, sessionId);
|
||||
await context.close();
|
||||
}, 60_000);
|
||||
|
||||
it('does not replay a full-history response, whatever geometry it reports', async () => {
|
||||
// A `full=1` body is linear scrollback closed by a RELATIVE cursor move,
|
||||
// which is relative precisely so the browser's row count need not match the
|
||||
// pane's. A mismatch there is not damage and a replay cannot repair it, so
|
||||
// the geometry comparison must not fire on it. This is the path that makes
|
||||
// the gate worth having: `_fullHistoryLoaded` is empty on the first select
|
||||
// of every non-shell session per page, so an ungated comparison would pull
|
||||
// the entire tmux scrollback a second time on every page load and every
|
||||
// first tab switch, for a session whose pane a desktop tab is holding too
|
||||
// tall to ever fit.
|
||||
context = await browser.newContext({ viewport: { width: 1280, height: 800 } });
|
||||
page = await context.newPage();
|
||||
const sessionId = await openSession(page);
|
||||
|
||||
const fetches = { n: 0, urls: [] as string[] };
|
||||
await stubTerminal(page, 200, fetches, { captureCols: WIDER_THAN_ANY_TERMINAL_COLS });
|
||||
await select(page, sessionId);
|
||||
|
||||
// Both dimensions are mismatched, so height alone is not what spares it.
|
||||
expect(await terminalRows(page)).toBeLessThan(200);
|
||||
expect(await terminalCols(page)).toBeLessThan(WIDER_THAN_ANY_TERMINAL_COLS);
|
||||
expect(fetches.n).toBe(1);
|
||||
expect(fetches.urls.filter((u) => u.includes('full=1'))).toHaveLength(1);
|
||||
|
||||
await closeSession(page, sessionId);
|
||||
await context.close();
|
||||
}, 60_000);
|
||||
|
||||
it('replays once per session, not once per tab switch, when it cannot converge', async () => {
|
||||
// `resizeRetry` caps the recursion inside ONE select and says nothing about
|
||||
// the next one, so a pane this browser cannot size reported the same
|
||||
// mismatch on every select and bought the same failed repair every time:
|
||||
// two fetches per tab switch for the life of the page. That is the case the
|
||||
// description calls "every time rather than occasionally", a phone whose
|
||||
// resize is declined while a desktop claim is live, and it is not the only
|
||||
// one — any pane Codeman cannot size lands there, a second tmux client
|
||||
// attached to it included. Each wasted pass costs another `capture-pane`,
|
||||
// which is `execSync` on the server's event loop, plus a reset and rewrite,
|
||||
// a discarded snapshot, and a dropped and reopened WebSocket.
|
||||
context = await browser.newContext({ viewport: { width: 1280, height: 800 } });
|
||||
page = await context.newPage();
|
||||
const sessionId = await openSession(page);
|
||||
|
||||
const fetches = { n: 0, urls: [] as string[] };
|
||||
const pane = { captureRows: 200, captureCols: 200 };
|
||||
await stubTerminalDynamic(page, fetches, pane);
|
||||
await consumeFullHistory(page, sessionId, fetches);
|
||||
|
||||
// First tab switch: one load, one replay, and the replay does not fit
|
||||
// either, which is the proof that this pane ignores the size it is given.
|
||||
await select(page, sessionId, { forceReload: true });
|
||||
expect(fetches.n).toBe(2);
|
||||
|
||||
// Every switch after it pays once. Unlatched this reads 4 then 6.
|
||||
await select(page, sessionId, { forceReload: true });
|
||||
expect(fetches.n).toBe(3);
|
||||
await select(page, sessionId, { forceReload: true });
|
||||
expect(fetches.n).toBe(4);
|
||||
|
||||
// The memo has to lift when the pane becomes sizeable again, or closing the
|
||||
// desktop tab that was holding it would leave this session permanently
|
||||
// unrepaired. A frame that fits clears it...
|
||||
pane.captureRows = 5;
|
||||
pane.captureCols = 40;
|
||||
await select(page, sessionId, { forceReload: true });
|
||||
expect(fetches.n).toBe(5);
|
||||
|
||||
// ...so the next genuine mismatch is diagnosed again.
|
||||
pane.captureRows = 200;
|
||||
pane.captureCols = 200;
|
||||
await select(page, sessionId, { forceReload: true });
|
||||
expect(fetches.n).toBe(7);
|
||||
|
||||
await closeSession(page, sessionId);
|
||||
await context.close();
|
||||
}, 60_000);
|
||||
|
||||
it('hands over text typed but not yet submitted before it replays', async () => {
|
||||
// On a touch device the characters the user has typed live ONLY in the
|
||||
// local-echo overlay until Enter; they have never reached the PTY. The
|
||||
// replay re-enters `selectSession` with `forceReload` on the session that
|
||||
// is still active, and that branch used to null `activeSessionId` before
|
||||
// `_cleanupPreviousSession` ran, so the flush there saw no session and the
|
||||
// unconditional `clear()` afterwards took the characters with it. Nothing
|
||||
// the user did triggered that: the replay fires on its own the moment a
|
||||
// tab switch finishes, which is exactly when someone typing into a
|
||||
// still-loading terminal has text in the overlay.
|
||||
context = await browser.newContext({ viewport: { width: 1280, height: 800 } });
|
||||
page = await context.newPage();
|
||||
const sessionId = await openSession(page);
|
||||
|
||||
const fetches = { n: 0, urls: [] as string[] };
|
||||
await stubTerminal(page, 200, fetches);
|
||||
await consumeFullHistory(page, sessionId, fetches);
|
||||
|
||||
// Headless chromium reports `isTouchDevice()` false even with `hasTouch`,
|
||||
// so the overlay would stay off and the whole case would pass vacuously.
|
||||
// The setting is what `_updateLocalEchoState()` reads, so it survives the
|
||||
// recompute that every select runs; the flag is forced too, for the window
|
||||
// before the next recompute. Record what crosses into the delivery layer,
|
||||
// which is the seam the text failed to cross.
|
||||
await page.evaluate(() => {
|
||||
const w = window as unknown as {
|
||||
app: {
|
||||
_localEchoEnabled: boolean;
|
||||
_sendInputAsync: (id: string, text: string, opts?: unknown) => void;
|
||||
terminal?: { focus: () => void };
|
||||
loadAppSettingsFromStorage: () => Record<string, unknown>;
|
||||
};
|
||||
__sentInputs: { id: string; text: string }[];
|
||||
};
|
||||
const settings = w.app.loadAppSettingsFromStorage();
|
||||
settings.localEchoEnabled = true;
|
||||
localStorage.setItem('codeman-app-settings', JSON.stringify(settings));
|
||||
w.app._localEchoEnabled = true;
|
||||
w.__sentInputs = [];
|
||||
const original = w.app._sendInputAsync.bind(w.app);
|
||||
w.app._sendInputAsync = (id: string, text: string, opts?: unknown) => {
|
||||
w.__sentInputs.push({ id, text });
|
||||
return original(id, text, opts);
|
||||
};
|
||||
w.app.terminal?.focus();
|
||||
});
|
||||
|
||||
await page.keyboard.type('hello-unsent');
|
||||
// The premise: the characters really are sitting in the overlay, unsent.
|
||||
// Without this the case would pass on a build where typing goes straight
|
||||
// to the PTY and there is nothing to lose.
|
||||
const pendingBefore = await page.evaluate(
|
||||
() =>
|
||||
(window as unknown as { app: { _localEchoOverlay?: { pendingText: string } } }).app._localEchoOverlay
|
||||
?.pendingText ?? ''
|
||||
);
|
||||
expect(pendingBefore).toBe('hello-unsent');
|
||||
|
||||
// The captured pane is taller than the terminal, so this select replays.
|
||||
await select(page, sessionId, { forceReload: true });
|
||||
expect(fetches.n).toBe(2);
|
||||
|
||||
const sent = await page.evaluate(
|
||||
() => (window as unknown as { __sentInputs: { id: string; text: string }[] }).__sentInputs
|
||||
);
|
||||
expect(sent.map((s) => s.text)).toContain('hello-unsent');
|
||||
expect(sent.find((s) => s.text === 'hello-unsent')?.id).toBe(sessionId);
|
||||
|
||||
await closeSession(page, sessionId);
|
||||
await context.close();
|
||||
}, 60_000);
|
||||
|
||||
it('does not replay a pane already at the size the client asked for', async () => {
|
||||
// `getTerminalDimensions()` floors at 40x10 while `fitAddon.fit()` does
|
||||
// not, so a viewport this small leaves the terminal shorter than the size
|
||||
// the client itself requests, and the pane obligingly draws at the floored
|
||||
// size. The captured height then exceeds the terminal's forever. A replay
|
||||
// cannot converge, because it re-requests the same floored size and
|
||||
// captures the same frame, so without the equality guard this retries on
|
||||
// every tab switch for the life of the page.
|
||||
context = await browser.newContext({ viewport: { width: 320, height: 200 } });
|
||||
page = await context.newPage();
|
||||
const sessionId = await openSession(page);
|
||||
|
||||
const fetches = { n: 0, urls: [] as string[] };
|
||||
await stubTerminalAtRequestedSize(page, fetches);
|
||||
await consumeFullHistory(page, sessionId, fetches);
|
||||
await select(page, sessionId, { forceReload: true });
|
||||
|
||||
// The premise: the floor really does bind here. Without this the case
|
||||
// would pass on any viewport, proving nothing.
|
||||
const requested = await page.evaluate(
|
||||
() =>
|
||||
(
|
||||
window as unknown as { app: { getTerminalDimensions?: () => { cols: number; rows: number } | null } }
|
||||
).app.getTerminalDimensions?.() ?? null
|
||||
);
|
||||
expect(requested).not.toBeNull();
|
||||
expect(requested!.rows).toBeGreaterThan(await terminalRows(page));
|
||||
|
||||
expect(fetches.n).toBe(1);
|
||||
|
||||
await closeSession(page, sessionId);
|
||||
await context.close();
|
||||
}, 60_000);
|
||||
});
|
||||
@@ -0,0 +1,180 @@
|
||||
/**
|
||||
* @fileoverview Output arriving after a pane capture survives the buffer load.
|
||||
*
|
||||
* `batchTerminalWrite` queues live terminal events while a buffer load runs,
|
||||
* and `_finishBufferLoad` discards that queue by default. That is right when
|
||||
* the loaded buffer is the server's accumulated byte history, which is current
|
||||
* up to the response. A tmux pane capture is current only up to CAPTURE time,
|
||||
* so anything arriving between the capture and the end of the chunked write is
|
||||
* queued and then dropped, with nothing scheduling a re-fetch.
|
||||
*
|
||||
* The queue now stamps each entry with its arrival time, and a capture load
|
||||
* replays the tail that arrived after the response headers. These drive the
|
||||
* real client in chromium: the event is injected from inside the response's
|
||||
* own `json()` call, which is the one place guaranteed to land after the
|
||||
* headers and before the chunked write.
|
||||
*
|
||||
* Port: 3256 (capture load window)
|
||||
*
|
||||
* Run: npx vitest run --config config/vitest.browser.config.ts test/capture-load-window.browser.test.ts
|
||||
*/
|
||||
|
||||
import { describe, it, expect, beforeAll, afterAll } from 'vitest';
|
||||
import { chromium, type Browser, type BrowserContext, type Page } from 'playwright';
|
||||
import { WebServer } from '../src/web/server.js';
|
||||
|
||||
const PORT = 3256;
|
||||
const BASE_URL = `http://localhost:${PORT}`;
|
||||
const MARKER = 'ARRIVED-AFTER-THE-CAPTURE';
|
||||
|
||||
let server: WebServer;
|
||||
let browser: Browser;
|
||||
|
||||
beforeAll(async () => {
|
||||
server = new WebServer(PORT, false, true); // testMode
|
||||
await server.start();
|
||||
browser = await chromium.launch({ headless: true });
|
||||
}, 60_000);
|
||||
|
||||
afterAll(async () => {
|
||||
await browser?.close();
|
||||
await server?.stop();
|
||||
}, 30_000);
|
||||
|
||||
/**
|
||||
* Select the session with the terminal fetch stubbed, injecting one live event
|
||||
* from inside `json()`. Returns how many terminal rows carry the marker, so a
|
||||
* flush that replays too much fails as loudly as one that replays nothing.
|
||||
*/
|
||||
async function runLoad(page: Page, sessionId: string, source: string): Promise<number> {
|
||||
return page.evaluate(
|
||||
async ({ sid, src, marker }) => {
|
||||
const app = (
|
||||
window as unknown as {
|
||||
app: {
|
||||
selectSession: (id: string, o?: object) => Promise<void>;
|
||||
_onSessionTerminal: (e: { id: string; data: string }) => void;
|
||||
terminal: {
|
||||
buffer: {
|
||||
active: {
|
||||
length: number;
|
||||
getLine: (i: number) => { translateToString: (t: boolean) => string } | undefined;
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
}
|
||||
).app;
|
||||
|
||||
const realFetch = window.fetch.bind(window);
|
||||
window.fetch = ((input: RequestInfo | URL, init?: RequestInit) => {
|
||||
const url = String(typeof input === 'string' ? input : ((input as Request).url ?? input));
|
||||
if (!url.includes('/terminal')) return realFetch(input as RequestInfo, init);
|
||||
return Promise.resolve({
|
||||
ok: true,
|
||||
status: 200,
|
||||
// `selectSession` timestamps the headers the moment this promise
|
||||
// resolves, then calls json(). Injecting here puts the event after
|
||||
// that timestamp and inside the load window, which is exactly the
|
||||
// gap a pane capture cannot cover.
|
||||
json: async () => {
|
||||
app._onSessionTerminal({ id: sid, data: `\r\n${marker}\r\n` });
|
||||
return {
|
||||
success: true,
|
||||
data: {
|
||||
terminalBuffer: '\x1b[1;1Hcaptured frame line one\r\n',
|
||||
status: 'idle',
|
||||
fullSize: 512,
|
||||
retainedBytes: 512,
|
||||
truncated: false,
|
||||
truncationReason: null,
|
||||
source: src,
|
||||
captureCols: 80,
|
||||
captureRows: 24,
|
||||
},
|
||||
};
|
||||
},
|
||||
}) as unknown as Promise<Response>;
|
||||
}) as typeof window.fetch;
|
||||
|
||||
try {
|
||||
await app.selectSession(sid);
|
||||
await new Promise((r) => setTimeout(r, 1200));
|
||||
const buf = app.terminal.buffer.active;
|
||||
let hits = 0;
|
||||
for (let i = 0; i < buf.length; i++) {
|
||||
if (buf.getLine(i)?.translateToString(true).includes(marker)) hits += 1;
|
||||
}
|
||||
return hits;
|
||||
} finally {
|
||||
window.fetch = realFetch;
|
||||
}
|
||||
},
|
||||
{ sid: sessionId, src: source, marker: MARKER }
|
||||
);
|
||||
}
|
||||
|
||||
async function openSession(page: Page): Promise<string> {
|
||||
await page.goto(BASE_URL, { waitUntil: 'domcontentloaded' });
|
||||
await page.waitForFunction(() => document.body.classList.contains('app-loaded'), { timeout: 10_000 });
|
||||
// xterm loads from /vendor, so the terminal appears a beat after the app.
|
||||
// Without it every buffer assertion below would throw rather than compare.
|
||||
await page.waitForFunction(() => (window as unknown as { app?: { terminal?: unknown } }).app?.terminal, null, {
|
||||
timeout: 30_000,
|
||||
});
|
||||
return page.evaluate(async () => {
|
||||
const res = await fetch('/api/sessions', {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ workingDir: '/tmp', name: 'capture-load-window-test' }),
|
||||
});
|
||||
const body = await res.json();
|
||||
return body.data?.session?.id ?? body.data?.id ?? body.id;
|
||||
});
|
||||
}
|
||||
|
||||
describe('output emitted during a capture load', () => {
|
||||
let context: BrowserContext;
|
||||
let page: Page;
|
||||
|
||||
afterAll(async () => {
|
||||
await context?.close();
|
||||
});
|
||||
|
||||
it('reaches the terminal exactly once when the buffer came from a pane capture', async () => {
|
||||
context = await browser.newContext({ viewport: { width: 1280, height: 800 } });
|
||||
page = await context.newPage();
|
||||
const sessionId = await openSession(page);
|
||||
expect(sessionId).toBeTruthy();
|
||||
|
||||
// Exactly once. The cutoff exists so the flush cannot also replay events the
|
||||
// payload already carried, which would double the output rather than heal it.
|
||||
expect(await runLoad(page, sessionId, 'mux-visible')).toBe(1);
|
||||
|
||||
await page.evaluate(
|
||||
(sid: string) => fetch(`/api/sessions/${sid}`, { method: 'DELETE' }).then(() => undefined),
|
||||
sessionId
|
||||
);
|
||||
await context.close();
|
||||
}, 60_000);
|
||||
|
||||
it('stays dropped when the buffer came from the accumulated byte history', async () => {
|
||||
// The byte history already contains everything up to the response, so
|
||||
// replaying the queue on top of it would duplicate the output — most
|
||||
// visibly Ink's cursor-up redraws. The discard has to survive this fix.
|
||||
context = await browser.newContext({ viewport: { width: 1280, height: 800 } });
|
||||
page = await context.newPage();
|
||||
const sessionId = await openSession(page);
|
||||
// Without this, a failed create passes the zero-hit assertion below
|
||||
// vacuously — nothing was loaded, so nothing was replayed.
|
||||
expect(sessionId).toBeTruthy();
|
||||
|
||||
expect(await runLoad(page, sessionId, 'history')).toBe(0);
|
||||
|
||||
await page.evaluate(
|
||||
(sid: string) => fetch(`/api/sessions/${sid}`, { method: 'DELETE' }).then(() => undefined),
|
||||
sessionId
|
||||
);
|
||||
await context.close();
|
||||
}, 60_000);
|
||||
});
|
||||
@@ -0,0 +1,368 @@
|
||||
/**
|
||||
* @fileoverview Tests for `refreshAllCustomModelHosts()`, the periodic
|
||||
* background sweep behind server.ts's "custom model endpoint re-discovery"
|
||||
* timer (docs/custom-model-endpoints-plan.md). Kept in its own file rather
|
||||
* than folded into test/routes/custom-model-routes.test.ts: that file's data
|
||||
* dir is shared across every test in it (one temp HOME per FILE, not per
|
||||
* test — test/setup.ts), and a sweep that walks every saved host would pick
|
||||
* up every host any other test in that file happened to create, making an
|
||||
* exact call-count or exact-host assertion meaningless. A dedicated file
|
||||
* gets its own clean temp HOME.
|
||||
*
|
||||
* Port: N/A (no server; drives readCustomModelHosts/writeCustomModelHosts
|
||||
* directly plus the mocked webviewFetch dispatcher).
|
||||
*/
|
||||
import { describe, it, expect, vi, beforeEach } from 'vitest';
|
||||
import { getDataDir } from '../src/config/instance.js';
|
||||
import { readCustomModelHosts, writeCustomModelHosts, type CustomModelHost } from '../src/custom-model-hosts.js';
|
||||
import { refreshAllCustomModelHosts } from '../src/web/routes/custom-model-routes.js';
|
||||
import { webviewFetch } from '../src/web/webview-egress.js';
|
||||
|
||||
vi.mock('../src/web/webview-egress.js', async () => {
|
||||
const actual = await vi.importActual<typeof import('../src/web/webview-egress.js')>('../src/web/webview-egress.js');
|
||||
return { ...actual, webviewFetch: vi.fn() };
|
||||
});
|
||||
|
||||
const fetchMock = vi.mocked(webviewFetch);
|
||||
|
||||
function host(overrides: Partial<CustomModelHost> & Pick<CustomModelHost, 'id' | 'baseUrl'>): CustomModelHost {
|
||||
return { label: overrides.id, ...overrides };
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
fetchMock.mockReset();
|
||||
});
|
||||
|
||||
describe('refreshAllCustomModelHosts (the periodic re-discovery sweep)', () => {
|
||||
it('refreshes every saved endpoint, best-effort — one unreachable host does not stop the others', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [
|
||||
host({ id: 'ok', baseUrl: 'http://localhost:8080' }),
|
||||
host({ id: 'down', baseUrl: 'http://localhost:8081' }),
|
||||
]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.href.includes('8081')) throw new TypeError('fetch failed', { cause: new Error('ECONNREFUSED') });
|
||||
return new Response(JSON.stringify({ data: [{ id: 'qwen3' }] }), { status: 200 });
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const hosts = await readCustomModelHosts(dir);
|
||||
const ok = hosts.find((h) => h.id === 'ok');
|
||||
const down = hosts.find((h) => h.id === 'down');
|
||||
expect(ok?.models).toEqual(['qwen3']);
|
||||
expect(ok?.lastDiscoveredAt).toBeTruthy();
|
||||
expect(down?.models ?? []).toEqual([]);
|
||||
expect(down?.lastDiscoveredAt).toBeFalsy();
|
||||
});
|
||||
|
||||
it('skips a host whose baseUrl is blocked, without making a request', async () => {
|
||||
const dir = getDataDir();
|
||||
// Written directly rather than through the POST route, which already
|
||||
// refuses this at save time — this simulates a record that pre-dates the
|
||||
// guard, or was hand-edited on disk. The sweep must not trust it either.
|
||||
await writeCustomModelHosts(dir, [host({ id: 'meta', baseUrl: 'http://169.254.169.254/' })]);
|
||||
|
||||
fetchMock.mockResolvedValue(new Response(JSON.stringify({ data: [{ id: 'x' }] }), { status: 200 }));
|
||||
await refreshAllCustomModelHosts();
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('drops a stale default and preserves lastDiscoveredAt semantics, same as manual discovery', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [
|
||||
host({ id: 'ep', baseUrl: 'http://localhost:8080', models: ['qwen3'], defaultModelId: 'qwen3' }),
|
||||
]);
|
||||
fetchMock.mockResolvedValue(new Response(JSON.stringify({ data: [{ id: 'llama3' }] }), { status: 200 }));
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.models).toEqual(['llama3']);
|
||||
expect(updated.defaultModelId).toBeUndefined();
|
||||
expect(updated.lastDiscoveredAt).toBeTruthy();
|
||||
});
|
||||
|
||||
it('keeps a default that is still present after the sweep', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [
|
||||
host({ id: 'ep', baseUrl: 'http://localhost:8080', models: ['qwen3'], defaultModelId: 'qwen3' }),
|
||||
]);
|
||||
fetchMock.mockResolvedValue(
|
||||
new Response(JSON.stringify({ data: [{ id: 'qwen3' }, { id: 'llama3' }] }), { status: 200 })
|
||||
);
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.defaultModelId).toBe('qwen3');
|
||||
});
|
||||
|
||||
it('does not resurrect an endpoint deleted while the sweep was in flight', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'deleted', baseUrl: 'http://localhost:8080' })]);
|
||||
|
||||
fetchMock.mockImplementation(async () => {
|
||||
// Simulate an admin deleting the endpoint between the sweep's fetch and
|
||||
// its read-modify-write — the delete must win, not be overwritten by a
|
||||
// refresh that started before it.
|
||||
const current = await readCustomModelHosts(dir);
|
||||
await writeCustomModelHosts(
|
||||
dir,
|
||||
current.filter((h) => h.id !== 'deleted')
|
||||
);
|
||||
return new Response(JSON.stringify({ data: [{ id: 'qwen3' }] }), { status: 200 });
|
||||
});
|
||||
|
||||
await expect(refreshAllCustomModelHosts()).resolves.toBeUndefined();
|
||||
const hosts = await readCustomModelHosts(dir);
|
||||
expect(hosts.find((h) => h.id === 'deleted')).toBeUndefined();
|
||||
});
|
||||
|
||||
it('leaves the store untouched when there are no saved endpoints at all', async () => {
|
||||
await expect(refreshAllCustomModelHosts()).resolves.toBeUndefined();
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
|
||||
describe('refreshAllCustomModelHosts: context-length enrichment (llama.cpp/llama-swap /props)', () => {
|
||||
it('probes /props?model= only for a model reported loaded, and stores its n_ctx', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
data: [
|
||||
{ id: 'loaded-model', status: { value: 'loaded' } },
|
||||
{ id: 'unloaded-model', status: { value: 'unloaded' } },
|
||||
],
|
||||
}),
|
||||
{ status: 200 }
|
||||
);
|
||||
}
|
||||
if (url.pathname === '/props') {
|
||||
// Must never be reached for the unloaded model — asserted below by call count.
|
||||
expect(url.searchParams.get('model')).toBe('loaded-model');
|
||||
return new Response(JSON.stringify({ n_ctx: 16384 }), { status: 200 });
|
||||
}
|
||||
throw new Error(`unexpected request: ${url.href}`);
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toEqual({ 'loaded-model': 16384 });
|
||||
const propsCalls = fetchMock.mock.calls.filter(([url]) => (url as URL).pathname === '/props');
|
||||
expect(propsCalls).toHaveLength(1);
|
||||
});
|
||||
|
||||
it('never probes /props at all when no entry mentions status — feature-detected, not assumed unloaded', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockResolvedValue(new Response(JSON.stringify({ data: [{ id: 'qwen3' }] }), { status: 200 }));
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1); // /v1/models only
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toBeUndefined();
|
||||
});
|
||||
|
||||
it('keeps a previously-learned context length for a model no longer loaded, drops it once the model disappears entirely', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [
|
||||
host({
|
||||
id: 'ep',
|
||||
baseUrl: 'http://localhost:8080',
|
||||
models: ['a', 'b'],
|
||||
modelContextLengths: { a: 8192, b: 4096 },
|
||||
}),
|
||||
]);
|
||||
// This round: 'a' is loaded (re-confirmed), 'b' is gone from the list entirely.
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 });
|
||||
}
|
||||
return new Response(JSON.stringify({ n_ctx: 8192 }), { status: 200 });
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toEqual({ a: 8192 });
|
||||
});
|
||||
|
||||
it('a failed /props probe for the loaded model is swallowed, leaving no context length rather than failing the sweep', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 });
|
||||
}
|
||||
return new Response('nope', { status: 500 });
|
||||
});
|
||||
|
||||
await expect(refreshAllCustomModelHosts()).resolves.toBeUndefined();
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toBeUndefined();
|
||||
});
|
||||
|
||||
it('prefers the REAL configured context size parsed from /running’s launch command over /props’s unreliable n_ctx', async () => {
|
||||
// Confirmed live: llama-swap launched a model with --fit-ctx 16384 (the real, working
|
||||
// limit — the actual server then refused a request over it), but /props reported
|
||||
// n_ctx: 154112 for the same model, well over what it would really accept. /props must
|
||||
// never be reached at all once the /running command parse already answered it.
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(JSON.stringify({ data: [{ id: 'qwen3.8-27b', status: { value: 'loaded' } }] }), {
|
||||
status: 200,
|
||||
});
|
||||
}
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
running: [
|
||||
{
|
||||
model: 'qwen3.8-27b',
|
||||
state: 'ready',
|
||||
cmd: 'llama-server -m /models/Qwen3.8-27B.gguf --flash-attn on --jinja --fit-ctx 16384 --host 0.0.0.0 --port 5840',
|
||||
},
|
||||
],
|
||||
}),
|
||||
{ status: 200 }
|
||||
);
|
||||
}
|
||||
if (url.pathname === '/props') throw new Error('must never be reached — the cmd parse already answered it');
|
||||
throw new Error(`unexpected request: ${url.href}`);
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toEqual({ 'qwen3.8-27b': 16384 });
|
||||
});
|
||||
|
||||
it('falls back to /props when /running has no cmd, or the cmd states no recognizable context flag', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 });
|
||||
}
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(
|
||||
JSON.stringify({ running: [{ model: 'a', state: 'ready', cmd: 'llama-server -m /models/a.gguf' }] }),
|
||||
{ status: 200 }
|
||||
);
|
||||
}
|
||||
if (url.pathname === '/props') return new Response(JSON.stringify({ n_ctx: 8192 }), { status: 200 });
|
||||
throw new Error(`unexpected request: ${url.href}`);
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toEqual({ a: 8192 });
|
||||
});
|
||||
|
||||
it('also recognizes a plain -c/--ctx-size flag, not just llama-swap’s own --fit-ctx', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(JSON.stringify({ data: [{ id: 'a', status: { value: 'loaded' } }] }), { status: 200 });
|
||||
}
|
||||
if (url.pathname === '/running') {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
running: [{ model: 'a', state: 'ready', cmd: 'llama-server -m /models/a.gguf --ctx-size 8192' }],
|
||||
}),
|
||||
{ status: 200 }
|
||||
);
|
||||
}
|
||||
throw new Error(`unexpected request: ${url.href}`); // /props must never be reached
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelContextLengths).toEqual({ a: 8192 });
|
||||
});
|
||||
});
|
||||
|
||||
describe('refreshAllCustomModelHosts: model-size enrichment (parsed from /v1/models description)', () => {
|
||||
it('parses a GB figure out of an auto-discovered model’s description', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockResolvedValue(
|
||||
new Response(
|
||||
JSON.stringify({
|
||||
data: [{ id: 'qwen3.8-27b', description: 'Auto-discovered 16.35 GB - parameters auto-fitted by llama.cpp' }],
|
||||
}),
|
||||
{ status: 200 }
|
||||
)
|
||||
);
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelSizesGB).toEqual({ 'qwen3.8-27b': 16.35 });
|
||||
});
|
||||
|
||||
it('gets no size at all for a hand-configured profile whose own description states none', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockResolvedValue(
|
||||
new Response(
|
||||
JSON.stringify({
|
||||
data: [{ id: 'big', description: 'General-purpose reasoning model, MoE CPU-offloaded. Default profile.' }],
|
||||
}),
|
||||
{ status: 200 }
|
||||
)
|
||||
);
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelSizesGB).toBeUndefined();
|
||||
});
|
||||
|
||||
it('populated regardless of loaded state — unlike context length, no /props probe is needed', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [host({ id: 'ep', baseUrl: 'http://localhost:8080' })]);
|
||||
fetchMock.mockImplementation(async (url: URL) => {
|
||||
if (url.pathname === '/v1/models') {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
data: [{ id: 'unloaded-model', description: 'Auto-discovered 4.91 GB - parameters auto-fitted' }],
|
||||
}),
|
||||
{ status: 200 }
|
||||
);
|
||||
}
|
||||
throw new Error(`unexpected request: ${url.href}`); // /props must never be reached for this
|
||||
});
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelSizesGB).toEqual({ 'unloaded-model': 4.91 });
|
||||
});
|
||||
|
||||
it('keeps a previously-learned size for a model still present, drops it once the model disappears entirely', async () => {
|
||||
const dir = getDataDir();
|
||||
await writeCustomModelHosts(dir, [
|
||||
host({ id: 'ep', baseUrl: 'http://localhost:8080', models: ['a', 'b'], modelSizesGB: { a: 8, b: 16 } }),
|
||||
]);
|
||||
fetchMock.mockResolvedValue(
|
||||
new Response(JSON.stringify({ data: [{ id: 'a', description: 'no GB figure here' }] }), { status: 200 })
|
||||
);
|
||||
|
||||
await refreshAllCustomModelHosts();
|
||||
|
||||
const [updated] = await readCustomModelHosts(dir);
|
||||
expect(updated.modelSizesGB).toEqual({ a: 8 }); // 'a' kept from before, 'b' dropped (gone from the list)
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,331 @@
|
||||
/**
|
||||
* @fileoverview Tests for the two custom-model IO-layer fixes on top of the pure builder
|
||||
* (docs/custom-model-endpoints-plan.md):
|
||||
*
|
||||
* 1. `contextLengthVar` — a discovered per-model context length reaches the actual
|
||||
* session env (CLAUDE_CODE_MAX_CONTEXT_TOKENS), so a CLI stops assuming a large
|
||||
* default window for an unrecognized custom model id and overflowing a much
|
||||
* smaller real one.
|
||||
* 2. `configDirVar` — an isolated, empty config directory is created and pointed at
|
||||
* (CLAUDE_CONFIG_DIR), so an injected API key never shares a directory with a
|
||||
* stored claude.ai OAuth session; `projects` is symlinked back into the real
|
||||
* config dir so the response viewer/subagent windows/Read My Mind keep working.
|
||||
*
|
||||
* Port: N/A (no server; filesystem-only, under a temp CODEMAN data dir from test/setup.ts).
|
||||
*/
|
||||
import { existsSync, lstatSync, mkdirSync, readFileSync, readdirSync, rmSync, writeFileSync } from 'node:fs';
|
||||
import { homedir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { afterEach, describe, expect, it } from 'vitest';
|
||||
import { getCli } from '../src/config/cli-registry/index.js';
|
||||
import { applyCustomModelInjection, customModelConfigDir } from '../src/custom-model-injection-apply.js';
|
||||
import type { CustomModelEndpoint } from '../src/custom-model-injection.js';
|
||||
|
||||
const endpoint: CustomModelEndpoint = {
|
||||
id: 'ep1',
|
||||
label: 'llama.cpp box',
|
||||
baseUrl: 'http://192.168.1.50:8080',
|
||||
apiKey: 'my-key',
|
||||
};
|
||||
|
||||
function entryOrThrow(id: string) {
|
||||
const entry = getCli(id);
|
||||
if (!entry) throw new Error(`missing CLI registry entry: ${id}`);
|
||||
return entry;
|
||||
}
|
||||
|
||||
const sessionsToClean: string[] = [];
|
||||
afterEach(() => {
|
||||
for (const id of sessionsToClean.splice(0)) rmSync(customModelConfigDir(id), { recursive: true, force: true });
|
||||
});
|
||||
|
||||
describe('applyCustomModelInjection: context length', () => {
|
||||
it('claude: passes a known context length through to CLAUDE_CODE_MAX_CONTEXT_TOKENS', () => {
|
||||
sessionsToClean.push('sess-ctx-1');
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', 'sess-ctx-1', 16384);
|
||||
expect(applied?.envOverrides.CLAUDE_CODE_MAX_CONTEXT_TOKENS).toBe('16384');
|
||||
expect(applied?.envKeys).toContain('CLAUDE_CODE_MAX_CONTEXT_TOKENS');
|
||||
});
|
||||
|
||||
it('claude: omits the var entirely when the context length is unknown', () => {
|
||||
sessionsToClean.push('sess-ctx-2');
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', 'sess-ctx-2');
|
||||
expect(applied?.envOverrides.CLAUDE_CODE_MAX_CONTEXT_TOKENS).toBeUndefined();
|
||||
});
|
||||
|
||||
it('deepseek: has no contextLengthVar declared, so a passed-in length is a no-op', () => {
|
||||
const applied = applyCustomModelInjection(entryOrThrow('deepseek'), endpoint, 'qwen3', 'sess-ctx-3', 16384);
|
||||
expect(Object.keys(applied?.envOverrides ?? {}).sort()).toEqual(['DEEPSEEK_API_KEY', 'DEEPSEEK_BASE_URL']);
|
||||
});
|
||||
});
|
||||
|
||||
describe('applyCustomModelInjection: CLAUDE_CONFIG_DIR isolation', () => {
|
||||
it('claude: creates an isolated config dir (no real credential/config files) and points CLAUDE_CONFIG_DIR at it', () => {
|
||||
const sessionId = 'sess-cfgdir-1';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
const expectedDir = customModelConfigDir(sessionId);
|
||||
expect(applied?.envOverrides.CLAUDE_CONFIG_DIR).toBe(expectedDir);
|
||||
expect(applied?.configDir).toBe(expectedDir);
|
||||
expect(existsSync(expectedDir)).toBe(true);
|
||||
// The trust-seed file, the skipFirstRunPrompts settings.json, and the projects link —
|
||||
// no real OAuth credential/config.
|
||||
const entries = readdirSync(expectedDir).filter((name) => name !== 'projects');
|
||||
expect(entries.sort()).toEqual(['.claude.json', 'settings.json']);
|
||||
});
|
||||
|
||||
it('claude: symlinks (or junctions) projects back to the real config dir so the response viewer keeps working', () => {
|
||||
const sessionId = 'sess-cfgdir-2';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
const link = join(applied!.configDir!, 'projects');
|
||||
// Best-effort: only assert the link exists if it was actually created (the real
|
||||
// ~/.claude/projects may not exist on a bare CI box, in which case linking is skipped).
|
||||
if (existsSync(join(homedir(), '.claude', 'projects'))) {
|
||||
expect(existsSync(link)).toBe(true);
|
||||
expect(lstatSync(link).isSymbolicLink() || lstatSync(link).isDirectory()).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
it('claude: re-applying to the same session is idempotent (boot-recovery re-apply)', () => {
|
||||
const sessionId = 'sess-cfgdir-3';
|
||||
sessionsToClean.push(sessionId);
|
||||
const first = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
const second = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
expect(second?.configDir).toBe(first?.configDir);
|
||||
expect(existsSync(first!.configDir!)).toBe(true);
|
||||
});
|
||||
|
||||
it('pi: configDir-kind CLIs are unaffected — no configDirVar concept for them', () => {
|
||||
const sessionId = 'sess-cfgdir-pi';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(entryOrThrow('pi'), endpoint, 'qwen3', sessionId);
|
||||
expect(applied?.envOverrides.HOME).toBe(customModelConfigDir(sessionId));
|
||||
});
|
||||
|
||||
it('deepseek: no configDirVar declared, so no config dir is created at all', () => {
|
||||
const sessionId = 'sess-cfgdir-deepseek';
|
||||
const applied = applyCustomModelInjection(entryOrThrow('deepseek'), endpoint, 'qwen3', sessionId);
|
||||
expect(applied?.configDir).toBeUndefined();
|
||||
expect(existsSync(customModelConfigDir(sessionId))).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('applyCustomModelInjection: apiKeyTrustFile (pre-approves the injected key)', () => {
|
||||
it('claude: seeds .claude.json so the "Detected a custom API key" prompt never fires', () => {
|
||||
const sessionId = 'sess-trust-1';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
customApiKeyResponses: { approved: string[]; rejected: string[] };
|
||||
};
|
||||
expect(written.customApiKeyResponses.approved).toEqual(['my-key']);
|
||||
expect(written.customApiKeyResponses.rejected).toEqual([]);
|
||||
});
|
||||
|
||||
// ⚠ Claude Code stores and looks up only the LAST 20 CHARACTERS of a key
|
||||
// (`key.trim().slice(-20)`, applied on both write and read), so seeding the whole key
|
||||
// never matches for a REAL one and the launch stops at the interactive "Detected a
|
||||
// custom API key" prompt whose default is "No (recommended)". Every other test here
|
||||
// uses a key shorter than 20 characters, where slice(-20) is the whole string and the
|
||||
// bug is invisible, which is exactly how it survived review.
|
||||
it('claude: seeds a REAL-length key in the truncated form the CLI actually matches on', () => {
|
||||
const sessionId = 'sess-trust-long';
|
||||
sessionsToClean.push(sessionId);
|
||||
const longKey = 'sk-or-v1-0123456789abcdef0123456789abcdef0123456789abcdef';
|
||||
expect(longKey.length).toBeGreaterThan(20);
|
||||
|
||||
const applied = applyCustomModelInjection(
|
||||
entryOrThrow('claude'),
|
||||
{ ...endpoint, apiKey: longKey },
|
||||
'qwen3',
|
||||
sessionId
|
||||
);
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
customApiKeyResponses: { approved: string[] };
|
||||
};
|
||||
|
||||
expect(written.customApiKeyResponses.approved).toEqual(['cdef0123456789abcdef']);
|
||||
expect(written.customApiKeyResponses.approved[0]).toHaveLength(20);
|
||||
// and the full credential is not written into this second file at all
|
||||
expect(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')).not.toContain(longKey);
|
||||
});
|
||||
|
||||
it('claude: falls back to the dummy key when the endpoint has none, and still seeds it', () => {
|
||||
const sessionId = 'sess-trust-2';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(
|
||||
entryOrThrow('claude'),
|
||||
{ ...endpoint, apiKey: undefined },
|
||||
'qwen3',
|
||||
sessionId
|
||||
);
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
customApiKeyResponses: { approved: string[] };
|
||||
};
|
||||
expect(written.customApiKeyResponses.approved).toEqual(['local-dummy-key']);
|
||||
});
|
||||
|
||||
it('claude: merges onto fields the CLI itself already wrote into the same isolated dir, never overwrites them', () => {
|
||||
const sessionId = 'sess-trust-3';
|
||||
sessionsToClean.push(sessionId);
|
||||
const configDir = customModelConfigDir(sessionId);
|
||||
mkdirSync(configDir, { recursive: true });
|
||||
writeFileSync(join(configDir, '.claude.json'), JSON.stringify({ userID: 'abc123', numStartups: 3 }));
|
||||
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
userID: string;
|
||||
numStartups: number;
|
||||
customApiKeyResponses: { approved: string[] };
|
||||
};
|
||||
expect(written.userID).toBe('abc123');
|
||||
expect(written.numStartups).toBe(3);
|
||||
expect(written.customApiKeyResponses.approved).toEqual(['my-key']);
|
||||
});
|
||||
|
||||
it('claude: a corrupt existing file is treated as absent rather than failing the apply', () => {
|
||||
const sessionId = 'sess-trust-4';
|
||||
sessionsToClean.push(sessionId);
|
||||
const configDir = customModelConfigDir(sessionId);
|
||||
mkdirSync(configDir, { recursive: true });
|
||||
writeFileSync(join(configDir, '.claude.json'), '{ not valid json');
|
||||
|
||||
expect(() => applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId)).not.toThrow();
|
||||
const written = JSON.parse(readFileSync(join(configDir, '.claude.json'), 'utf8')) as {
|
||||
customApiKeyResponses: { approved: string[] };
|
||||
};
|
||||
expect(written.customApiKeyResponses.approved).toEqual(['my-key']);
|
||||
});
|
||||
|
||||
it('claude: re-approving the same key does not duplicate it in the approved list', () => {
|
||||
const sessionId = 'sess-trust-5';
|
||||
sessionsToClean.push(sessionId);
|
||||
applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
const second = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'llama3', sessionId);
|
||||
const written = JSON.parse(readFileSync(join(second!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
customApiKeyResponses: { approved: string[] };
|
||||
};
|
||||
expect(written.customApiKeyResponses.approved).toEqual(['my-key']);
|
||||
});
|
||||
|
||||
it('opencode: has no apiKeyTrustFile declared (no configDirVar at all), nothing is seeded', () => {
|
||||
const sessionId = 'sess-trust-opencode';
|
||||
const applied = applyCustomModelInjection(entryOrThrow('opencode'), endpoint, 'qwen3', sessionId);
|
||||
expect(applied?.configDir).toBeUndefined();
|
||||
expect(existsSync(customModelConfigDir(sessionId))).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("applyCustomModelInjection: skipFirstRunPrompts (an isolated dir replays claude's whole first-run sequence)", () => {
|
||||
it("claude: seeds hasCompletedOnboarding and this session's own project trust into .claude.json", () => {
|
||||
const sessionId = 'sess-firstrun-1';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(
|
||||
entryOrThrow('claude'),
|
||||
endpoint,
|
||||
'qwen3',
|
||||
sessionId,
|
||||
undefined,
|
||||
'/home/user/myproject'
|
||||
);
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
hasCompletedOnboarding: boolean;
|
||||
projects: Record<string, { hasTrustDialogAccepted: boolean }>;
|
||||
};
|
||||
expect(written.hasCompletedOnboarding).toBe(true);
|
||||
expect(written.projects['/home/user/myproject'].hasTrustDialogAccepted).toBe(true);
|
||||
});
|
||||
|
||||
it('claude: seeds skipDangerousModePermissionPrompt into settings.json', () => {
|
||||
const sessionId = 'sess-firstrun-2';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, 'settings.json'), 'utf8')) as {
|
||||
skipDangerousModePermissionPrompt: boolean;
|
||||
};
|
||||
expect(written.skipDangerousModePermissionPrompt).toBe(true);
|
||||
});
|
||||
|
||||
it('claude: with no workingDir given (boot recovery), hasCompletedOnboarding/settings still seed, but no project entry is added', () => {
|
||||
const sessionId = 'sess-firstrun-3';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId);
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
hasCompletedOnboarding: boolean;
|
||||
projects?: Record<string, unknown>;
|
||||
};
|
||||
expect(written.hasCompletedOnboarding).toBeUndefined();
|
||||
expect(written.projects).toBeUndefined();
|
||||
});
|
||||
|
||||
it("claude: merges onto an existing project entry's other fields rather than overwriting them", () => {
|
||||
const sessionId = 'sess-firstrun-4';
|
||||
sessionsToClean.push(sessionId);
|
||||
const configDir = customModelConfigDir(sessionId);
|
||||
mkdirSync(configDir, { recursive: true });
|
||||
writeFileSync(
|
||||
join(configDir, '.claude.json'),
|
||||
JSON.stringify({ projects: { '/home/user/myproject': { allowedTools: ['Bash'] } } })
|
||||
);
|
||||
|
||||
const applied = applyCustomModelInjection(
|
||||
entryOrThrow('claude'),
|
||||
endpoint,
|
||||
'qwen3',
|
||||
sessionId,
|
||||
undefined,
|
||||
'/home/user/myproject'
|
||||
);
|
||||
|
||||
const written = JSON.parse(readFileSync(join(applied!.configDir!, '.claude.json'), 'utf8')) as {
|
||||
projects: Record<string, { allowedTools: string[]; hasTrustDialogAccepted: boolean }>;
|
||||
};
|
||||
expect(written.projects['/home/user/myproject'].allowedTools).toEqual(['Bash']);
|
||||
expect(written.projects['/home/user/myproject'].hasTrustDialogAccepted).toBe(true);
|
||||
});
|
||||
|
||||
it('claude: a corrupt existing settings.json is treated as absent rather than failing the apply', () => {
|
||||
const sessionId = 'sess-firstrun-5';
|
||||
sessionsToClean.push(sessionId);
|
||||
const configDir = customModelConfigDir(sessionId);
|
||||
mkdirSync(configDir, { recursive: true });
|
||||
writeFileSync(join(configDir, 'settings.json'), '{ not valid json');
|
||||
|
||||
expect(() => applyCustomModelInjection(entryOrThrow('claude'), endpoint, 'qwen3', sessionId)).not.toThrow();
|
||||
const written = JSON.parse(readFileSync(join(configDir, 'settings.json'), 'utf8')) as {
|
||||
skipDangerousModePermissionPrompt: boolean;
|
||||
};
|
||||
expect(written.skipDangerousModePermissionPrompt).toBe(true);
|
||||
});
|
||||
|
||||
it('pi: has no skipFirstRunPrompts concept (no apiKeyTrustFile either) — nothing beyond its own config file', () => {
|
||||
const sessionId = 'sess-firstrun-pi';
|
||||
sessionsToClean.push(sessionId);
|
||||
const applied = applyCustomModelInjection(
|
||||
entryOrThrow('pi'),
|
||||
endpoint,
|
||||
'qwen3',
|
||||
sessionId,
|
||||
undefined,
|
||||
'/home/user/myproject'
|
||||
);
|
||||
const entries = readdirSync(applied!.configDir!);
|
||||
expect(entries).not.toContain('settings.json');
|
||||
});
|
||||
});
|
||||
|
||||
describe('applyCustomModelInjection: pre-existing behavior unaffected', () => {
|
||||
it('opencode: still returns a plain env-kind result with no configDir', () => {
|
||||
const sessionId = 'sess-opencode-1';
|
||||
const applied = applyCustomModelInjection(entryOrThrow('opencode'), endpoint, 'qwen3', sessionId);
|
||||
expect(applied?.configDir).toBeUndefined();
|
||||
expect(applied?.envOverrides.OPENCODE_CONFIG_CONTENT).toBeTruthy();
|
||||
});
|
||||
|
||||
it('antigravity: still undefined (unsupported)', () => {
|
||||
const applied = applyCustomModelInjection(entryOrThrow('antigravity'), endpoint, 'qwen3', 'sess-agy-1');
|
||||
expect(applied).toBeUndefined();
|
||||
});
|
||||
});
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user